{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ESAUWQAYBSHU325IYOXZTBY47L","short_pith_number":"pith:ESAUWQAY","schema_version":"1.0","canonical_sha256":"24814b40180c8f4deba8c3af99871cfae2c3c3acd83bdc0d3dc25a97d0646aaa","source":{"kind":"arxiv","id":"2406.11303","version":1},"attestation_state":"computed","paper":{"title":"VideoVista: A Versatile Benchmark for Video Understanding and Reasoning","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Baotian Hu, Haoyuan Shi, Longyue Wang, Min Zhang, Xinyu Chen, Yunxin Li","submitted_at":"2024-06-17T08:09:00Z","abstract_excerpt":"Despite significant breakthroughs in video analysis driven by the rapid development of large multimodal models (LMMs), there remains a lack of a versatile evaluation benchmark to comprehensively assess these models' performance in video understanding and reasoning. To address this, we present VideoVista, a video QA benchmark that integrates challenges across diverse content categories, durations, and abilities. Specifically, VideoVista comprises 25,000 questions derived from 3,400 videos spanning 14 categories (e.g., Howto, Film, and Entertainment) with durations ranging from a few seconds to "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.11303","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-06-17T08:09:00Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"cbbad0cfa65f07356db53a6454836ddbbcb28753321f70facb3cd336597cdaaf","abstract_canon_sha256":"fbdf2f6465cdded8af0dc9cc3c031ef1e5b814e6863c6bf0b477d4944bd025b7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:32:51.603350Z","signature_b64":"GT7YvOqJJrTy5iBzugN9lt893I7PNJqU2KahvuIR4VkSAXSOD0sJ5LoZBQ/lVVfZ2oyJTGZzjck6cOmVj0UeDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"24814b40180c8f4deba8c3af99871cfae2c3c3acd83bdc0d3dc25a97d0646aaa","last_reissued_at":"2026-07-05T08:32:51.602855Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:32:51.602855Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VideoVista: A Versatile Benchmark for Video Understanding and Reasoning","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Baotian Hu, Haoyuan Shi, Longyue Wang, Min Zhang, Xinyu Chen, Yunxin Li","submitted_at":"2024-06-17T08:09:00Z","abstract_excerpt":"Despite significant breakthroughs in video analysis driven by the rapid development of large multimodal models (LMMs), there remains a lack of a versatile evaluation benchmark to comprehensively assess these models' performance in video understanding and reasoning. To address this, we present VideoVista, a video QA benchmark that integrates challenges across diverse content categories, durations, and abilities. Specifically, VideoVista comprises 25,000 questions derived from 3,400 videos spanning 14 categories (e.g., Howto, Film, and Entertainment) with durations ranging from a few seconds to "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.11303","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.11303/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.11303","created_at":"2026-07-05T08:32:51.602912+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.11303v1","created_at":"2026-07-05T08:32:51.602912+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.11303","created_at":"2026-07-05T08:32:51.602912+00:00"},{"alias_kind":"pith_short_12","alias_value":"ESAUWQAYBSHU","created_at":"2026-07-05T08:32:51.602912+00:00"},{"alias_kind":"pith_short_16","alias_value":"ESAUWQAYBSHU325I","created_at":"2026-07-05T08:32:51.602912+00:00"},{"alias_kind":"pith_short_8","alias_value":"ESAUWQAY","created_at":"2026-07-05T08:32:51.602912+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.06338","citing_title":"StoryVideoQA: Scaling Deep Video Understanding with a Large-Scale, Multi-Genre and Auto-Generated Dataset","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03635","citing_title":"VidMsg: A Benchmark for Implicit Message Inference in Short Videos","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28593","citing_title":"Animation2Code: Evaluating Temporal Visual Reasoning in Video-to-Code Generation","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2501.02955","citing_title":"MotionBench: Benchmarking and Improving Fine-grained Video Motion Understanding for Vision Language Models","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15342","citing_title":"Minerva-Ego: Spatiotemporal Hints for Egocentric Video Understanding","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2506.05425","citing_title":"SIV-Bench: A Video Benchmark for Social Interaction Understanding and Reasoning","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2603.27259","citing_title":"Seeing the Scene Matters: Revealing Forgetting in Video Understanding Models with a Scene-Aware Long-Video Benchmark","ref_index":25,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ESAUWQAYBSHU325IYOXZTBY47L","json":"https://pith.science/pith/ESAUWQAYBSHU325IYOXZTBY47L.json","graph_json":"https://pith.science/api/pith-number/ESAUWQAYBSHU325IYOXZTBY47L/graph.json","events_json":"https://pith.science/api/pith-number/ESAUWQAYBSHU325IYOXZTBY47L/events.json","paper":"https://pith.science/paper/ESAUWQAY"},"agent_actions":{"view_html":"https://pith.science/pith/ESAUWQAYBSHU325IYOXZTBY47L","download_json":"https://pith.science/pith/ESAUWQAYBSHU325IYOXZTBY47L.json","view_paper":"https://pith.science/paper/ESAUWQAY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.11303&json=true","fetch_graph":"https://pith.science/api/pith-number/ESAUWQAYBSHU325IYOXZTBY47L/graph.json","fetch_events":"https://pith.science/api/pith-number/ESAUWQAYBSHU325IYOXZTBY47L/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ESAUWQAYBSHU325IYOXZTBY47L/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ESAUWQAYBSHU325IYOXZTBY47L/action/storage_attestation","attest_author":"https://pith.science/pith/ESAUWQAYBSHU325IYOXZTBY47L/action/author_attestation","sign_citation":"https://pith.science/pith/ESAUWQAYBSHU325IYOXZTBY47L/action/citation_signature","submit_replication":"https://pith.science/pith/ESAUWQAYBSHU325IYOXZTBY47L/action/replication_record"}},"created_at":"2026-07-05T08:32:51.602912+00:00","updated_at":"2026-07-05T08:32:51.602912+00:00"}