{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:LOCLVP4S6GV67QDRXR7N7OLZT6","short_pith_number":"pith:LOCLVP4S","schema_version":"1.0","canonical_sha256":"5b84babf92f1abefc071bc7edfb9799fa6c4d4a7b66d301b7b98cfa5a4ec3d9e","source":{"kind":"arxiv","id":"2506.09987","version":1},"attestation_state":"computed","paper":{"title":"A Shortcut-aware Video-QA Benchmark for Physical Understanding via Minimal Video Pairs","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Benno Krojer, Candace Ross, Koustuv Sinha, Mahmoud Assran, Mojtaba Komeili, Nicolas Ballas, Quentin Garrido","submitted_at":"2025-06-11T17:57:32Z","abstract_excerpt":"Existing benchmarks for assessing the spatio-temporal understanding and reasoning abilities of video language models are susceptible to score inflation due to the presence of shortcut solutions based on superficial visual or textual cues. This paper mitigates the challenges in accurately assessing model performance by introducing the Minimal Video Pairs (MVP) benchmark, a simple shortcut-aware video QA benchmark for assessing the physical understanding of video language models. The benchmark is comprised of 55K high-quality multiple-choice video QA examples focusing on physical world understan"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.09987","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2025-06-11T17:57:32Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"b1a3d36cc1a73de2a5abcfb3d6ed79aec8fa0d9292f50411fe41d04c088d7df2","abstract_canon_sha256":"e60a97bcc313feb1370c6180bc6e018b38c7fa87ce548bb19d57e5494b152b86"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:20:00.214061Z","signature_b64":"M2odDiq6543A997VFPxXUXOq8bRyOxeWu5MjHupMuOvE1Jbat+iG45hCkmLUOguihRXpyuCQ8fI2uJneVvHnCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5b84babf92f1abefc071bc7edfb9799fa6c4d4a7b66d301b7b98cfa5a4ec3d9e","last_reissued_at":"2026-07-05T11:20:00.213459Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:20:00.213459Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Shortcut-aware Video-QA Benchmark for Physical Understanding via Minimal Video Pairs","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Benno Krojer, Candace Ross, Koustuv Sinha, Mahmoud Assran, Mojtaba Komeili, Nicolas Ballas, Quentin Garrido","submitted_at":"2025-06-11T17:57:32Z","abstract_excerpt":"Existing benchmarks for assessing the spatio-temporal understanding and reasoning abilities of video language models are susceptible to score inflation due to the presence of shortcut solutions based on superficial visual or textual cues. This paper mitigates the challenges in accurately assessing model performance by introducing the Minimal Video Pairs (MVP) benchmark, a simple shortcut-aware video QA benchmark for assessing the physical understanding of video language models. The benchmark is comprised of 55K high-quality multiple-choice video QA examples focusing on physical world understan"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.09987","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.09987/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.09987","created_at":"2026-07-05T11:20:00.213547+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.09987v1","created_at":"2026-07-05T11:20:00.213547+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.09987","created_at":"2026-07-05T11:20:00.213547+00:00"},{"alias_kind":"pith_short_12","alias_value":"LOCLVP4S6GV6","created_at":"2026-07-05T11:20:00.213547+00:00"},{"alias_kind":"pith_short_16","alias_value":"LOCLVP4S6GV67QDR","created_at":"2026-07-05T11:20:00.213547+00:00"},{"alias_kind":"pith_short_8","alias_value":"LOCLVP4S","created_at":"2026-07-05T11:20:00.213547+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.15032","citing_title":"How Should World Models Be Evaluated for Embodied Decision-Making? A Decision-Making-Centric Position","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2602.13294","citing_title":"VisPhyWorld: Probing Physical Reasoning via Code-Driven Video Reconstruction","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21988","citing_title":"Learning Spatiotemporal Sensitivity in Video LLMs via Counterfactual Reinforcement Learning","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22570","citing_title":"VGenST-Bench: A Benchmark for Spatio-Temporal Reasoning via Active Video Synthesis","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09904","citing_title":"TOC-Bench: A Temporal Object Consistency Benchmark for Video Large Language Models","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09904","citing_title":"TOC-Bench: A Temporal Object Consistency Benchmark for Video Large Language Models","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LOCLVP4S6GV67QDRXR7N7OLZT6","json":"https://pith.science/pith/LOCLVP4S6GV67QDRXR7N7OLZT6.json","graph_json":"https://pith.science/api/pith-number/LOCLVP4S6GV67QDRXR7N7OLZT6/graph.json","events_json":"https://pith.science/api/pith-number/LOCLVP4S6GV67QDRXR7N7OLZT6/events.json","paper":"https://pith.science/paper/LOCLVP4S"},"agent_actions":{"view_html":"https://pith.science/pith/LOCLVP4S6GV67QDRXR7N7OLZT6","download_json":"https://pith.science/pith/LOCLVP4S6GV67QDRXR7N7OLZT6.json","view_paper":"https://pith.science/paper/LOCLVP4S","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.09987&json=true","fetch_graph":"https://pith.science/api/pith-number/LOCLVP4S6GV67QDRXR7N7OLZT6/graph.json","fetch_events":"https://pith.science/api/pith-number/LOCLVP4S6GV67QDRXR7N7OLZT6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LOCLVP4S6GV67QDRXR7N7OLZT6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LOCLVP4S6GV67QDRXR7N7OLZT6/action/storage_attestation","attest_author":"https://pith.science/pith/LOCLVP4S6GV67QDRXR7N7OLZT6/action/author_attestation","sign_citation":"https://pith.science/pith/LOCLVP4S6GV67QDRXR7N7OLZT6/action/citation_signature","submit_replication":"https://pith.science/pith/LOCLVP4S6GV67QDRXR7N7OLZT6/action/replication_record"}},"created_at":"2026-07-05T11:20:00.213547+00:00","updated_at":"2026-07-05T11:20:00.213547+00:00"}