{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:W7KJSUVVZSJ4CL323QE4VLF7JW","short_pith_number":"pith:W7KJSUVV","schema_version":"1.0","canonical_sha256":"b7d49952b5cc93c12f7adc09caacbf4d8f33c102a31f52bc24ddf2c604f13b58","source":{"kind":"arxiv","id":"2308.09126","version":1},"attestation_state":"computed","paper":{"title":"EgoSchema: A Diagnostic Benchmark for Very Long-form Video Language Understanding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Jitendra Malik, Karttikeya Mangalam, Raiymbek Akshulakov","submitted_at":"2023-08-17T17:59:59Z","abstract_excerpt":"We introduce EgoSchema, a very long-form video question-answering dataset, and benchmark to evaluate long video understanding capabilities of modern vision and language systems. Derived from Ego4D, EgoSchema consists of over 5000 human curated multiple choice question answer pairs, spanning over 250 hours of real video data, covering a very broad range of natural human activity and behavior. For each question, EgoSchema requires the correct answer to be selected between five given options based on a three-minute-long video clip. While some prior works have proposed video datasets with long cli"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.09126","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-08-17T17:59:59Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"cb9a65b03d4208f041814ba3bb177d7e37ae50c3caab677bbd97ea5580af0d8d","abstract_canon_sha256":"41bee9e443a498efe0e5d856461e9733daea7649d1f78418b2044c6d44d099ce"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:42:28.061454Z","signature_b64":"+1ZfE3KGZuzol3Rtj4/OJW6SF8/+eZrLhgpQX2sk3qJr9BpZBg/4/dazonFisz0Eu3Cnl47lCVrb9EOenK0LDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b7d49952b5cc93c12f7adc09caacbf4d8f33c102a31f52bc24ddf2c604f13b58","last_reissued_at":"2026-07-05T06:42:28.060981Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:42:28.060981Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"EgoSchema: A Diagnostic Benchmark for Very Long-form Video Language Understanding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Jitendra Malik, Karttikeya Mangalam, Raiymbek Akshulakov","submitted_at":"2023-08-17T17:59:59Z","abstract_excerpt":"We introduce EgoSchema, a very long-form video question-answering dataset, and benchmark to evaluate long video understanding capabilities of modern vision and language systems. Derived from Ego4D, EgoSchema consists of over 5000 human curated multiple choice question answer pairs, spanning over 250 hours of real video data, covering a very broad range of natural human activity and behavior. For each question, EgoSchema requires the correct answer to be selected between five given options based on a three-minute-long video clip. While some prior works have proposed video datasets with long cli"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.09126","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.09126/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.09126","created_at":"2026-07-05T06:42:28.061037+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.09126v1","created_at":"2026-07-05T06:42:28.061037+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.09126","created_at":"2026-07-05T06:42:28.061037+00:00"},{"alias_kind":"pith_short_12","alias_value":"W7KJSUVVZSJ4","created_at":"2026-07-05T06:42:28.061037+00:00"},{"alias_kind":"pith_short_16","alias_value":"W7KJSUVVZSJ4CL32","created_at":"2026-07-05T06:42:28.061037+00:00"},{"alias_kind":"pith_short_8","alias_value":"W7KJSUVV","created_at":"2026-07-05T06:42:28.061037+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":15,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24422","citing_title":"EgoSAT: A Comprehensive Benchmark of Egocentric Streaming Interaction Understanding","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04773","citing_title":"NextMotionQA: Benchmarking and Judging Human Motion Understanding with Vision-Language Models","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07639","citing_title":"MOSS-Video-Preview: Toward Real-Time Video Understanding via Cross-Attention","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24456","citing_title":"EgoProx: Evaluating MLLMs on Egocentric 3D Proximity Reasoning Across a Cognitive Hierarchy","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00616","citing_title":"Pause and Think: A Dataset and Benchmark for Video-Grounded Assistive Action Suggestion","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30220","citing_title":"From Accuracy to Visual Dependence: Auditing and Filtering Modality Collapse in Traffic VideoQA","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30673","citing_title":"TeachObs: A Human-Validated Benchmark for Multimodal Teaching Observation and Model Evaluation","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00616","citing_title":"Pause and Think: A Dataset and Benchmark for Video-Grounded Assistive Action Suggestion","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00825","citing_title":"SuperMemory-VQA: An Egocentric Visual Question-Answering Benchmark for Long-Horizon Memory","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19559","citing_title":"EgoCoT-Bench: Benchmarking Grounded and Verifiable Operation-Centric Chain of Thought Reasoning for MLLMs","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2311.17005","citing_title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2601.14724","citing_title":"HERMES: KV Cache as Hierarchical Memory for Efficient Streaming Video Understanding","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2603.20180","citing_title":"Adaptive Greedy Frame Selection for Long Video Understanding","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06747","citing_title":"HumanNet: Scaling Human-centric Video Learning to One Million Hours","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05225","citing_title":"MACS: Modality-Aware Capacity Scaling for Efficient Multimodal MoE Inference","ref_index":35,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/W7KJSUVVZSJ4CL323QE4VLF7JW","json":"https://pith.science/pith/W7KJSUVVZSJ4CL323QE4VLF7JW.json","graph_json":"https://pith.science/api/pith-number/W7KJSUVVZSJ4CL323QE4VLF7JW/graph.json","events_json":"https://pith.science/api/pith-number/W7KJSUVVZSJ4CL323QE4VLF7JW/events.json","paper":"https://pith.science/paper/W7KJSUVV"},"agent_actions":{"view_html":"https://pith.science/pith/W7KJSUVVZSJ4CL323QE4VLF7JW","download_json":"https://pith.science/pith/W7KJSUVVZSJ4CL323QE4VLF7JW.json","view_paper":"https://pith.science/paper/W7KJSUVV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.09126&json=true","fetch_graph":"https://pith.science/api/pith-number/W7KJSUVVZSJ4CL323QE4VLF7JW/graph.json","fetch_events":"https://pith.science/api/pith-number/W7KJSUVVZSJ4CL323QE4VLF7JW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/W7KJSUVVZSJ4CL323QE4VLF7JW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/W7KJSUVVZSJ4CL323QE4VLF7JW/action/storage_attestation","attest_author":"https://pith.science/pith/W7KJSUVVZSJ4CL323QE4VLF7JW/action/author_attestation","sign_citation":"https://pith.science/pith/W7KJSUVVZSJ4CL323QE4VLF7JW/action/citation_signature","submit_replication":"https://pith.science/pith/W7KJSUVVZSJ4CL323QE4VLF7JW/action/replication_record"}},"created_at":"2026-07-05T06:42:28.061037+00:00","updated_at":"2026-07-05T06:42:28.061037+00:00"}