{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ZY7HIXH7W2PBJ4FJQPPUTO7FOY","short_pith_number":"pith:ZY7HIXH7","schema_version":"1.0","canonical_sha256":"ce3e745cffb69e14f0a983df49bbe576038a5f218ee4db774ae783c318432577","source":{"kind":"arxiv","id":"2506.04280","version":1},"attestation_state":"computed","paper":{"title":"Evaluating MLLMs with Multimodal Multi-image Reasoning Benchmark","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Binrui Xu, Fangxiang Feng, Lei Ren, Lisheng Gong, Mingjie Zhan, Mingxiang Chen, Shiqi Zhong, Siyu Ren, Tianshuo Zhou, Wei Chen, Xiangchao Meng, Xiaojie Wang, Yanlin Li, Yuxin Zhang, Zhiyuan Huang, Ziming Cheng, Zuhe Song","submitted_at":"2025-06-04T04:21:32Z","abstract_excerpt":"With enhanced capabilities and widespread applications, Multimodal Large Language Models (MLLMs) are increasingly required to process and reason over multiple images simultaneously. However, existing MLLM benchmarks focus either on single-image visual reasoning or on multi-image understanding tasks with only final-answer evaluation, leaving the reasoning capabilities of MLLMs over multi-image inputs largely underexplored. To address this gap, we introduce the $\\textbf{Multimodal Multi-image Reasoning Benchmark (MMRB)}$, the first benchmark designed to evaluate structured visual reasoning acros"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.04280","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-06-04T04:21:32Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"76a0c04766b271273aaaf9f0090f03a9dd4687bfae61d2912897179666f73e23","abstract_canon_sha256":"63ffc178514e4a2348ba85eb415941f78db72c6e73c3b7851d0a13bb474905a3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:16:11.609010Z","signature_b64":"Gn/ECempF01O5CQQg2ji8ieLMrq4J8t14SCpZ+rLK5r4ESii8fHIHQ1U1xccYh1QpxC+i4PPa4/Jvjb1PyKiCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ce3e745cffb69e14f0a983df49bbe576038a5f218ee4db774ae783c318432577","last_reissued_at":"2026-07-05T11:16:11.608398Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:16:11.608398Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evaluating MLLMs with Multimodal Multi-image Reasoning Benchmark","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Binrui Xu, Fangxiang Feng, Lei Ren, Lisheng Gong, Mingjie Zhan, Mingxiang Chen, Shiqi Zhong, Siyu Ren, Tianshuo Zhou, Wei Chen, Xiangchao Meng, Xiaojie Wang, Yanlin Li, Yuxin Zhang, Zhiyuan Huang, Ziming Cheng, Zuhe Song","submitted_at":"2025-06-04T04:21:32Z","abstract_excerpt":"With enhanced capabilities and widespread applications, Multimodal Large Language Models (MLLMs) are increasingly required to process and reason over multiple images simultaneously. However, existing MLLM benchmarks focus either on single-image visual reasoning or on multi-image understanding tasks with only final-answer evaluation, leaving the reasoning capabilities of MLLMs over multi-image inputs largely underexplored. To address this gap, we introduce the $\\textbf{Multimodal Multi-image Reasoning Benchmark (MMRB)}$, the first benchmark designed to evaluate structured visual reasoning acros"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.04280","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.04280/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.04280","created_at":"2026-07-05T11:16:11.608469+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.04280v1","created_at":"2026-07-05T11:16:11.608469+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.04280","created_at":"2026-07-05T11:16:11.608469+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZY7HIXH7W2PB","created_at":"2026-07-05T11:16:11.608469+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZY7HIXH7W2PBJ4FJ","created_at":"2026-07-05T11:16:11.608469+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZY7HIXH7","created_at":"2026-07-05T11:16:11.608469+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.00148","citing_title":"StemBind: When MLLMs Get Lost Between Rules and Instances in Abstract Visual Reasoning","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.00799","citing_title":"Multimodal Language Models Cannot Spot Spatial Inconsistencies","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20806","citing_title":"OMIBench: Benchmarking Olympiad-Level Multi-Image Reasoning in Large Vision-Language Model","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19544","citing_title":"DT2IT-MRM: Debiased Preference Construction and Iterative Training for Multimodal Reward Modeling","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZY7HIXH7W2PBJ4FJQPPUTO7FOY","json":"https://pith.science/pith/ZY7HIXH7W2PBJ4FJQPPUTO7FOY.json","graph_json":"https://pith.science/api/pith-number/ZY7HIXH7W2PBJ4FJQPPUTO7FOY/graph.json","events_json":"https://pith.science/api/pith-number/ZY7HIXH7W2PBJ4FJQPPUTO7FOY/events.json","paper":"https://pith.science/paper/ZY7HIXH7"},"agent_actions":{"view_html":"https://pith.science/pith/ZY7HIXH7W2PBJ4FJQPPUTO7FOY","download_json":"https://pith.science/pith/ZY7HIXH7W2PBJ4FJQPPUTO7FOY.json","view_paper":"https://pith.science/paper/ZY7HIXH7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.04280&json=true","fetch_graph":"https://pith.science/api/pith-number/ZY7HIXH7W2PBJ4FJQPPUTO7FOY/graph.json","fetch_events":"https://pith.science/api/pith-number/ZY7HIXH7W2PBJ4FJQPPUTO7FOY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZY7HIXH7W2PBJ4FJQPPUTO7FOY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZY7HIXH7W2PBJ4FJQPPUTO7FOY/action/storage_attestation","attest_author":"https://pith.science/pith/ZY7HIXH7W2PBJ4FJQPPUTO7FOY/action/author_attestation","sign_citation":"https://pith.science/pith/ZY7HIXH7W2PBJ4FJQPPUTO7FOY/action/citation_signature","submit_replication":"https://pith.science/pith/ZY7HIXH7W2PBJ4FJQPPUTO7FOY/action/replication_record"}},"created_at":"2026-07-05T11:16:11.608469+00:00","updated_at":"2026-07-05T11:16:11.608469+00:00"}