{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:Q4JXSBEJETG2OC3UMU2DGR6E6I","short_pith_number":"pith:Q4JXSBEJ","schema_version":"1.0","canonical_sha256":"871379048924cda70b7465343347c4f209c1265562ab9b61ab6852b0eee4716b","source":{"kind":"arxiv","id":"2406.09175","version":1},"attestation_state":"computed","paper":{"title":"ReMI: A Dataset for Reasoning with Multiple Images","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Ahmed Qureshi, Ankit Anand, Bahare Fatemi, Dee Guo, Fangyu Liu, Ishita Dasgupta, Mehran Kazemi, Nishanth Dikkala, Petar Devic, Pranjal Awasthi, Sreenivas Gollapudi","submitted_at":"2024-06-13T14:37:04Z","abstract_excerpt":"With the continuous advancement of large language models (LLMs), it is essential to create new benchmarks to effectively evaluate their expanding capabilities and identify areas for improvement. This work focuses on multi-image reasoning, an emerging capability in state-of-the-art LLMs. We introduce ReMI, a dataset designed to assess LLMs' ability to Reason with Multiple Images. This dataset encompasses a diverse range of tasks, spanning various reasoning domains such as math, physics, logic, code, table/chart understanding, and spatial and temporal reasoning. It also covers a broad spectrum o"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.09175","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-06-13T14:37:04Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"c52b0ea082dcf4c22f6ecc6ae93b58c82532de11a4717592025a2de632d7ed2a","abstract_canon_sha256":"b75098cda1d4f76ef05546ab5687956d3d021606547bd40de64bdd8262869c84"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:31:29.601532Z","signature_b64":"+3u4tNep0ntpw+m0E0mSSBz7Ob5ectSVJQrC4NPBQwAjhw3v+DxcXDv3F79MD70PB42tl98dTJ7MxvdImNeeCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"871379048924cda70b7465343347c4f209c1265562ab9b61ab6852b0eee4716b","last_reissued_at":"2026-07-05T08:31:29.601001Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:31:29.601001Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ReMI: A Dataset for Reasoning with Multiple Images","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Ahmed Qureshi, Ankit Anand, Bahare Fatemi, Dee Guo, Fangyu Liu, Ishita Dasgupta, Mehran Kazemi, Nishanth Dikkala, Petar Devic, Pranjal Awasthi, Sreenivas Gollapudi","submitted_at":"2024-06-13T14:37:04Z","abstract_excerpt":"With the continuous advancement of large language models (LLMs), it is essential to create new benchmarks to effectively evaluate their expanding capabilities and identify areas for improvement. This work focuses on multi-image reasoning, an emerging capability in state-of-the-art LLMs. We introduce ReMI, a dataset designed to assess LLMs' ability to Reason with Multiple Images. This dataset encompasses a diverse range of tasks, spanning various reasoning domains such as math, physics, logic, code, table/chart understanding, and spatial and temporal reasoning. It also covers a broad spectrum o"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.09175","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.09175/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.09175","created_at":"2026-07-05T08:31:29.601070+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.09175v1","created_at":"2026-07-05T08:31:29.601070+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.09175","created_at":"2026-07-05T08:31:29.601070+00:00"},{"alias_kind":"pith_short_12","alias_value":"Q4JXSBEJETG2","created_at":"2026-07-05T08:31:29.601070+00:00"},{"alias_kind":"pith_short_16","alias_value":"Q4JXSBEJETG2OC3U","created_at":"2026-07-05T08:31:29.601070+00:00"},{"alias_kind":"pith_short_8","alias_value":"Q4JXSBEJ","created_at":"2026-07-05T08:31:29.601070+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25343","citing_title":"Invoice Haystack: Benchmarking Document Retrieval and Visual Question Answering Under Strong Visual Homogeneity","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2606.25343","citing_title":"Invoice Haystack: Benchmarking Document Retrieval and Visual Question Answering Under Strong Visual Homogeneity","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2503.19786","citing_title":"Gemma 3 Technical Report","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2407.07895","citing_title":"LLaVA-NeXT-Interleave: Tackling Multi-image, Video, and 3D in Large Multimodal Models","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Q4JXSBEJETG2OC3UMU2DGR6E6I","json":"https://pith.science/pith/Q4JXSBEJETG2OC3UMU2DGR6E6I.json","graph_json":"https://pith.science/api/pith-number/Q4JXSBEJETG2OC3UMU2DGR6E6I/graph.json","events_json":"https://pith.science/api/pith-number/Q4JXSBEJETG2OC3UMU2DGR6E6I/events.json","paper":"https://pith.science/paper/Q4JXSBEJ"},"agent_actions":{"view_html":"https://pith.science/pith/Q4JXSBEJETG2OC3UMU2DGR6E6I","download_json":"https://pith.science/pith/Q4JXSBEJETG2OC3UMU2DGR6E6I.json","view_paper":"https://pith.science/paper/Q4JXSBEJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.09175&json=true","fetch_graph":"https://pith.science/api/pith-number/Q4JXSBEJETG2OC3UMU2DGR6E6I/graph.json","fetch_events":"https://pith.science/api/pith-number/Q4JXSBEJETG2OC3UMU2DGR6E6I/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Q4JXSBEJETG2OC3UMU2DGR6E6I/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Q4JXSBEJETG2OC3UMU2DGR6E6I/action/storage_attestation","attest_author":"https://pith.science/pith/Q4JXSBEJETG2OC3UMU2DGR6E6I/action/author_attestation","sign_citation":"https://pith.science/pith/Q4JXSBEJETG2OC3UMU2DGR6E6I/action/citation_signature","submit_replication":"https://pith.science/pith/Q4JXSBEJETG2OC3UMU2DGR6E6I/action/replication_record"}},"created_at":"2026-07-05T08:31:29.601070+00:00","updated_at":"2026-07-05T08:31:29.601070+00:00"}