{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:6C4GTIVNKOCUFFB3T6EHDPFNB7","short_pith_number":"pith:6C4GTIVN","schema_version":"1.0","canonical_sha256":"f0b869a2ad538542943b9f8871bcad0ff99db47547a67b7b12aafed5f42f9c2b","source":{"kind":"arxiv","id":"2501.12380","version":1},"attestation_state":"computed","paper":{"title":"MMVU: Measuring Expert-Level Multi-Discipline Video Understanding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Arman Cohan, Chengye Wang, Chen Zhao, Chuhan Li, Guo Gan, Haowei Zhang, Junyang Song, Lujing Xie, Tongyan Hu, Weifeng Pan, Weiyuan Chen, Xiangru Tang, Yilun Zhao, Yitao Long, Yixin Liu, Zhenwen Liang, Zhijian Xu, Zhiyuan Hu, Ziyao Shangguan","submitted_at":"2025-01-21T18:56:18Z","abstract_excerpt":"We introduce MMVU, a comprehensive expert-level, multi-discipline benchmark for evaluating foundation models in video understanding. MMVU includes 3,000 expert-annotated questions spanning 27 subjects across four core disciplines: Science, Healthcare, Humanities & Social Sciences, and Engineering. Compared to prior benchmarks, MMVU features three key advancements. First, it challenges models to apply domain-specific knowledge and perform expert-level reasoning to analyze specialized-domain videos, moving beyond the basic visual perception typically assessed in current video benchmarks. Second,"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.12380","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-01-21T18:56:18Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"df710c6e8b85a2ed864ef4d8ff944f5765fe27275aacaefd2de6e8aaae5c15ba","abstract_canon_sha256":"3eb077d139bd7ed7b4ac205586a4160ccee215c3e90b2c0684864beb279446c8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:03:35.281114Z","signature_b64":"i3Qfbwzjs0CPt9y+JgHG6FfdqwyrGNwwoGX5a/YvtAkMTIRNt8pBmbpIv93SHTwQAvSOOo3KT0ERSSSObtUZCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f0b869a2ad538542943b9f8871bcad0ff99db47547a67b7b12aafed5f42f9c2b","last_reissued_at":"2026-07-05T10:03:35.280622Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:03:35.280622Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MMVU: Measuring Expert-Level Multi-Discipline Video Understanding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Arman Cohan, Chengye Wang, Chen Zhao, Chuhan Li, Guo Gan, Haowei Zhang, Junyang Song, Lujing Xie, Tongyan Hu, Weifeng Pan, Weiyuan Chen, Xiangru Tang, Yilun Zhao, Yitao Long, Yixin Liu, Zhenwen Liang, Zhijian Xu, Zhiyuan Hu, Ziyao Shangguan","submitted_at":"2025-01-21T18:56:18Z","abstract_excerpt":"We introduce MMVU, a comprehensive expert-level, multi-discipline benchmark for evaluating foundation models in video understanding. MMVU includes 3,000 expert-annotated questions spanning 27 subjects across four core disciplines: Science, Healthcare, Humanities & Social Sciences, and Engineering. Compared to prior benchmarks, MMVU features three key advancements. First, it challenges models to apply domain-specific knowledge and perform expert-level reasoning to analyze specialized-domain videos, moving beyond the basic visual perception typically assessed in current video benchmarks. Second,"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.12380","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.12380/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.12380","created_at":"2026-07-05T10:03:35.280681+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.12380v1","created_at":"2026-07-05T10:03:35.280681+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.12380","created_at":"2026-07-05T10:03:35.280681+00:00"},{"alias_kind":"pith_short_12","alias_value":"6C4GTIVNKOCU","created_at":"2026-07-05T10:03:35.280681+00:00"},{"alias_kind":"pith_short_16","alias_value":"6C4GTIVNKOCUFFB3","created_at":"2026-07-05T10:03:35.280681+00:00"},{"alias_kind":"pith_short_8","alias_value":"6C4GTIVN","created_at":"2026-07-05T10:03:35.280681+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18216","citing_title":"Zone of Proximal Policy Optimization: Teacher in Prompts, Not Gradients","ref_index":151,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05736","citing_title":"VTI-CoT: Visual-Textual Interleaved Chain of Thought for Video Reasoning","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2502.02871","citing_title":"Position: Multimodal Large Language Models Can Significantly Advance Scientific Reasoning","ref_index":258,"is_internal_anchor":false},{"citing_arxiv_id":"2502.13923","citing_title":"Qwen2.5-VL Technical Report","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19559","citing_title":"EgoCoT-Bench: Benchmarking Grounded and Verifiable Operation-Centric Chain of Thought Reasoning for MLLMs","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2505.21374","citing_title":"Video-Holmes: Can MLLM Think Like Holmes for Complex Video Reasoning?","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27083","citing_title":"Co-Evolving Policy Distillation","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2503.21776","citing_title":"Video-R1: Reinforcing Video Reasoning in MLLMs","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20473","citing_title":"Video-ToC: Video Tree-of-Cue Reasoning","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2505.07062","citing_title":"Seed1.5-VL Technical Report","ref_index":176,"is_internal_anchor":false},{"citing_arxiv_id":"2602.02276","citing_title":"Kimi K2.5: Visual Agentic Intelligence","ref_index":82,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16893","citing_title":"EasyVideoR1: Easier RL for Video Understanding","ref_index":54,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6C4GTIVNKOCUFFB3T6EHDPFNB7","json":"https://pith.science/pith/6C4GTIVNKOCUFFB3T6EHDPFNB7.json","graph_json":"https://pith.science/api/pith-number/6C4GTIVNKOCUFFB3T6EHDPFNB7/graph.json","events_json":"https://pith.science/api/pith-number/6C4GTIVNKOCUFFB3T6EHDPFNB7/events.json","paper":"https://pith.science/paper/6C4GTIVN"},"agent_actions":{"view_html":"https://pith.science/pith/6C4GTIVNKOCUFFB3T6EHDPFNB7","download_json":"https://pith.science/pith/6C4GTIVNKOCUFFB3T6EHDPFNB7.json","view_paper":"https://pith.science/paper/6C4GTIVN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.12380&json=true","fetch_graph":"https://pith.science/api/pith-number/6C4GTIVNKOCUFFB3T6EHDPFNB7/graph.json","fetch_events":"https://pith.science/api/pith-number/6C4GTIVNKOCUFFB3T6EHDPFNB7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6C4GTIVNKOCUFFB3T6EHDPFNB7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6C4GTIVNKOCUFFB3T6EHDPFNB7/action/storage_attestation","attest_author":"https://pith.science/pith/6C4GTIVNKOCUFFB3T6EHDPFNB7/action/author_attestation","sign_citation":"https://pith.science/pith/6C4GTIVNKOCUFFB3T6EHDPFNB7/action/citation_signature","submit_replication":"https://pith.science/pith/6C4GTIVNKOCUFFB3T6EHDPFNB7/action/replication_record"}},"created_at":"2026-07-05T10:03:35.280681+00:00","updated_at":"2026-07-05T10:03:35.280681+00:00"}