{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:FT2QKMZXMBQSL56JDCKRRO7FL6","short_pith_number":"pith:FT2QKMZX","schema_version":"1.0","canonical_sha256":"2cf5053337606125f7c9189518bbe55f9640e8745f54278e416a0d80a7a80fd3","source":{"kind":"arxiv","id":"2504.05040","version":2},"attestation_state":"computed","paper":{"title":"InstructionBench: An Instructional Video Understanding Benchmark","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Haiwan Wei, Lin Ma, Wei Ke, Xiaohan Lan, Yitian Yuan","submitted_at":"2025-04-07T13:05:09Z","abstract_excerpt":"Despite progress in video large language models (Video-LLMs), research on instructional video understanding, crucial for enhancing access to instructional content, remains insufficient. To address this, we introduce InstructionBench, an Instructional video understanding Benchmark, which challenges models' advanced temporal reasoning within instructional videos characterized by their strict step-by-step flow. Employing GPT-4, we formulate Q&A pairs in open-ended and multiple-choice formats to assess both Coarse-Grained event-level and Fine-Grained object-level reasoning. Our filtering strategie"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.05040","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-04-07T13:05:09Z","cross_cats_sorted":[],"title_canon_sha256":"ea78f1f824b9e7fd5fc4250359d4ed5f1906dd2f04f71db8786f2ede9fafda6c","abstract_canon_sha256":"e7ee97d8ee5a07de1333513c61279e62f4cc8a644bbe62c7cf6a4b3050884e61"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:29:10.197840Z","signature_b64":"4P9aTTAn2VEuqov7fKvYxw4+FoqyLNn7S3Df/VWTJY7SgDkI09OK+2GLbZDEirWEUu5og6gMchaH0YrwGo1NAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2cf5053337606125f7c9189518bbe55f9640e8745f54278e416a0d80a7a80fd3","last_reissued_at":"2026-07-05T11:29:10.197395Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:29:10.197395Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"InstructionBench: An Instructional Video Understanding Benchmark","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Haiwan Wei, Lin Ma, Wei Ke, Xiaohan Lan, Yitian Yuan","submitted_at":"2025-04-07T13:05:09Z","abstract_excerpt":"Despite progress in video large language models (Video-LLMs), research on instructional video understanding, crucial for enhancing access to instructional content, remains insufficient. To address this, we introduce InstructionBench, an Instructional video understanding Benchmark, which challenges models' advanced temporal reasoning within instructional videos characterized by their strict step-by-step flow. Employing GPT-4, we formulate Q&A pairs in open-ended and multiple-choice formats to assess both Coarse-Grained event-level and Fine-Grained object-level reasoning. Our filtering strategie"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.05040","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.05040/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.05040","created_at":"2026-07-05T11:29:10.197455+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.05040v2","created_at":"2026-07-05T11:29:10.197455+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.05040","created_at":"2026-07-05T11:29:10.197455+00:00"},{"alias_kind":"pith_short_12","alias_value":"FT2QKMZXMBQS","created_at":"2026-07-05T11:29:10.197455+00:00"},{"alias_kind":"pith_short_16","alias_value":"FT2QKMZXMBQSL56J","created_at":"2026-07-05T11:29:10.197455+00:00"},{"alias_kind":"pith_short_8","alias_value":"FT2QKMZX","created_at":"2026-07-05T11:29:10.197455+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07433","citing_title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","ref_index":254,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FT2QKMZXMBQSL56JDCKRRO7FL6","json":"https://pith.science/pith/FT2QKMZXMBQSL56JDCKRRO7FL6.json","graph_json":"https://pith.science/api/pith-number/FT2QKMZXMBQSL56JDCKRRO7FL6/graph.json","events_json":"https://pith.science/api/pith-number/FT2QKMZXMBQSL56JDCKRRO7FL6/events.json","paper":"https://pith.science/paper/FT2QKMZX"},"agent_actions":{"view_html":"https://pith.science/pith/FT2QKMZXMBQSL56JDCKRRO7FL6","download_json":"https://pith.science/pith/FT2QKMZXMBQSL56JDCKRRO7FL6.json","view_paper":"https://pith.science/paper/FT2QKMZX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.05040&json=true","fetch_graph":"https://pith.science/api/pith-number/FT2QKMZXMBQSL56JDCKRRO7FL6/graph.json","fetch_events":"https://pith.science/api/pith-number/FT2QKMZXMBQSL56JDCKRRO7FL6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FT2QKMZXMBQSL56JDCKRRO7FL6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FT2QKMZXMBQSL56JDCKRRO7FL6/action/storage_attestation","attest_author":"https://pith.science/pith/FT2QKMZXMBQSL56JDCKRRO7FL6/action/author_attestation","sign_citation":"https://pith.science/pith/FT2QKMZXMBQSL56JDCKRRO7FL6/action/citation_signature","submit_replication":"https://pith.science/pith/FT2QKMZXMBQSL56JDCKRRO7FL6/action/replication_record"}},"created_at":"2026-07-05T11:29:10.197455+00:00","updated_at":"2026-07-05T11:29:10.197455+00:00"}