{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:VRPCEEKMQC3IVJ6QJZS2ZQJDMN","short_pith_number":"pith:VRPCEEKM","schema_version":"1.0","canonical_sha256":"ac5e22114c80b68aa7d04e65acc123635ec0ec73b132d9150a788da5d408ead6","source":{"kind":"arxiv","id":"2506.04141","version":1},"attestation_state":"computed","paper":{"title":"MMR-V: What's Left Unsaid? A Benchmark for Multimodal Deep Reasoning in Videos","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Hongbang Yuan, Jiachun Li, Jun Zhao, Kang Liu, Kejian Zhu, Pengfei Cao, Shangqing Tu, Yubo Chen, Zhuoran Jin","submitted_at":"2025-06-04T16:33:41Z","abstract_excerpt":"The sequential structure of videos poses a challenge to the ability of multimodal large language models (MLLMs) to locate multi-frame evidence and conduct multimodal reasoning. However, existing video benchmarks mainly focus on understanding tasks, which only require models to match frames mentioned in the question (hereafter referred to as \"question frame\") and perceive a few adjacent frames. To address this gap, we propose MMR-V: A Benchmark for Multimodal Deep Reasoning in Videos. The benchmark is characterized by the following features. (1) Long-range, multi-frame reasoning: Models are req"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.04141","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-06-04T16:33:41Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"9fc6ffe8116f6bd7b7124703715d0996200420a6401568a24c86f0c817d60c06","abstract_canon_sha256":"9d4078c25a580f21cf8650bf101e516dd181e237ef8d5ab853be1ad7976976cd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:15:57.865040Z","signature_b64":"rVSWdEmQyv2xVSAtSMEi2zJapG1nsuLaoAax9R8cAXORYDNsBtnK3Tyg9uaV40uXGhfcqkgu6Uqpm+rzEMmHBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ac5e22114c80b68aa7d04e65acc123635ec0ec73b132d9150a788da5d408ead6","last_reissued_at":"2026-07-05T11:15:57.864514Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:15:57.864514Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MMR-V: What's Left Unsaid? A Benchmark for Multimodal Deep Reasoning in Videos","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Hongbang Yuan, Jiachun Li, Jun Zhao, Kang Liu, Kejian Zhu, Pengfei Cao, Shangqing Tu, Yubo Chen, Zhuoran Jin","submitted_at":"2025-06-04T16:33:41Z","abstract_excerpt":"The sequential structure of videos poses a challenge to the ability of multimodal large language models (MLLMs) to locate multi-frame evidence and conduct multimodal reasoning. However, existing video benchmarks mainly focus on understanding tasks, which only require models to match frames mentioned in the question (hereafter referred to as \"question frame\") and perceive a few adjacent frames. To address this gap, we propose MMR-V: A Benchmark for Multimodal Deep Reasoning in Videos. The benchmark is characterized by the following features. (1) Long-range, multi-frame reasoning: Models are req"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.04141","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.04141/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.04141","created_at":"2026-07-05T11:15:57.864568+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.04141v1","created_at":"2026-07-05T11:15:57.864568+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.04141","created_at":"2026-07-05T11:15:57.864568+00:00"},{"alias_kind":"pith_short_12","alias_value":"VRPCEEKMQC3I","created_at":"2026-07-05T11:15:57.864568+00:00"},{"alias_kind":"pith_short_16","alias_value":"VRPCEEKMQC3IVJ6Q","created_at":"2026-07-05T11:15:57.864568+00:00"},{"alias_kind":"pith_short_8","alias_value":"VRPCEEKM","created_at":"2026-07-05T11:15:57.864568+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2606.12191","citing_title":"Agentic Environment Engineering for Large Language Models: A Survey of Environment Modeling, Synthesis, Evaluation, and Application","ref_index":53,"is_internal_anchor":true},{"citing_arxiv_id":"2605.21988","citing_title":"Learning Spatiotemporal Sensitivity in Video LLMs via Counterfactual Reinforcement Learning","ref_index":53,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VRPCEEKMQC3IVJ6QJZS2ZQJDMN","json":"https://pith.science/pith/VRPCEEKMQC3IVJ6QJZS2ZQJDMN.json","graph_json":"https://pith.science/api/pith-number/VRPCEEKMQC3IVJ6QJZS2ZQJDMN/graph.json","events_json":"https://pith.science/api/pith-number/VRPCEEKMQC3IVJ6QJZS2ZQJDMN/events.json","paper":"https://pith.science/paper/VRPCEEKM"},"agent_actions":{"view_html":"https://pith.science/pith/VRPCEEKMQC3IVJ6QJZS2ZQJDMN","download_json":"https://pith.science/pith/VRPCEEKMQC3IVJ6QJZS2ZQJDMN.json","view_paper":"https://pith.science/paper/VRPCEEKM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.04141&json=true","fetch_graph":"https://pith.science/api/pith-number/VRPCEEKMQC3IVJ6QJZS2ZQJDMN/graph.json","fetch_events":"https://pith.science/api/pith-number/VRPCEEKMQC3IVJ6QJZS2ZQJDMN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VRPCEEKMQC3IVJ6QJZS2ZQJDMN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VRPCEEKMQC3IVJ6QJZS2ZQJDMN/action/storage_attestation","attest_author":"https://pith.science/pith/VRPCEEKMQC3IVJ6QJZS2ZQJDMN/action/author_attestation","sign_citation":"https://pith.science/pith/VRPCEEKMQC3IVJ6QJZS2ZQJDMN/action/citation_signature","submit_replication":"https://pith.science/pith/VRPCEEKMQC3IVJ6QJZS2ZQJDMN/action/replication_record"}},"created_at":"2026-07-05T11:15:57.864568+00:00","updated_at":"2026-07-05T11:15:57.864568+00:00"}