{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:T7EF4SUOFTLLSQDQNLC2BWIURX","short_pith_number":"pith:T7EF4SUO","canonical_record":{"source":{"id":"2406.04264","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-06-06T17:09:32Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"18a5bf14850be2ac26f321769b1dc07c5948b5228a893e1a555f8f75b88e6ccd","abstract_canon_sha256":"41148e2d78f4d31b4e397c1a844d4b6f5ce24671e5f8064a2655bd442342726a"},"schema_version":"1.0"},"canonical_sha256":"9fc85e4a8e2cd6b940706ac5a0d9148de724540951e474f6b45f20ab24059c3a","source":{"kind":"arxiv","id":"2406.04264","version":3},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2406.04264","created_at":"2026-05-17T23:39:21Z"},{"alias_kind":"arxiv_version","alias_value":"2406.04264v3","created_at":"2026-05-17T23:39:21Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.04264","created_at":"2026-05-17T23:39:21Z"},{"alias_kind":"pith_short_12","alias_value":"T7EF4SUOFTLL","created_at":"2026-05-18T12:33:37Z"},{"alias_kind":"pith_short_16","alias_value":"T7EF4SUOFTLLSQDQ","created_at":"2026-05-18T12:33:37Z"},{"alias_kind":"pith_short_8","alias_value":"T7EF4SUO","created_at":"2026-05-18T12:33:37Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:T7EF4SUOFTLLSQDQNLC2BWIURX","target":"record","payload":{"canonical_record":{"source":{"id":"2406.04264","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-06-06T17:09:32Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"18a5bf14850be2ac26f321769b1dc07c5948b5228a893e1a555f8f75b88e6ccd","abstract_canon_sha256":"41148e2d78f4d31b4e397c1a844d4b6f5ce24671e5f8064a2655bd442342726a"},"schema_version":"1.0"},"canonical_sha256":"9fc85e4a8e2cd6b940706ac5a0d9148de724540951e474f6b45f20ab24059c3a","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-17T23:39:21.958888Z","signature_b64":"KTb9S0ZdoT5X7XkVuO1YvtcJ6lef3HAFS3b7z6E2zvRhYDsraWyyuyAtIBuVa27fzEb/GiJfZqTZnpQUzBx6BA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9fc85e4a8e2cd6b940706ac5a0d9148de724540951e474f6b45f20ab24059c3a","last_reissued_at":"2026-05-17T23:39:21.958186Z","signature_status":"signed_v1","first_computed_at":"2026-05-17T23:39:21.958186Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2406.04264","source_version":3,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-17T23:39:21Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"V62Yh7307G5S9YL8cMTxy4Qsclp9dZt2/pIqUA7abQ5baZk2QpTmsjzW+yE8pCInRN+m/fXEMKGazKKB/fEHDg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-07-26T04:48:06.974411Z"},"content_sha256":"1feb9bf7005aa08e00307218042e2ea0c81e7c9bc18f713ff9c838ba500e0ca9","schema_version":"1.0","event_id":"sha256:1feb9bf7005aa08e00307218042e2ea0c81e7c9bc18f713ff9c838ba500e0ca9"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:T7EF4SUOFTLLSQDQNLC2BWIURX","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"MLVU: Benchmarking Multi-task Long Video Understanding","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"MLVU benchmark shows current multimodal models struggle with most long video tasks and degrade sharply on longer clips.","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Boya Wu, Bo Zhang, Bo Zhao, Junjie Zhou, Minghao Qin, Shitao Xiao, Tiejun Huang, Xi Yang, Yan Shu, Yongping Xiong, Zheng Liu, Zhengyang Liang","submitted_at":"2024-06-06T17:09:32Z","abstract_excerpt":"The evaluation of Long Video Understanding (LVU) performance poses an important but challenging research problem. Despite previous efforts, the existing video understanding benchmarks are severely constrained by several issues, especially the insufficient lengths of videos, a lack of diversity in video types and evaluation tasks, and the inappropriateness for evaluating LVU performances. To address the above problems, we propose a new benchmark called MLVU (Multi-task Long Video Understanding Benchmark) for the comprehensive and in-depth evaluation of LVU. MLVU presents the following critical "},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"The empirical study with 23 latest MLLMs reveals significant room for improvement in today's technique, as all existing methods struggle with most of the evaluation tasks and exhibit severe performance degradation when handling longer videos.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"The chosen video lengths, genres, and tasks in MLVU sufficiently represent the core challenges of real-world long video understanding and that performance on these tasks generalizes beyond the benchmark.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"MLVU is a new benchmark for long video understanding that uses extended videos across diverse genres and multi-task evaluations, revealing that current MLLMs struggle significantly and degrade sharply with longer durations.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"MLVU benchmark shows current multimodal models struggle with most long video tasks and degrade sharply on longer clips.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"c886ba5b9cf2cbdfb5ba0c78b921da150b529af8bb67e4b124e6a3446d2abb83"},"source":{"id":"2406.04264","kind":"arxiv","version":3},"verdict":{"id":"49f26cf4-8e6d-424a-beee-af7ca6b1887f","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-14T19:49:55.465952Z","strongest_claim":"The empirical study with 23 latest MLLMs reveals significant room for improvement in today's technique, as all existing methods struggle with most of the evaluation tasks and exhibit severe performance degradation when handling longer videos.","one_line_summary":"MLVU is a new benchmark for long video understanding that uses extended videos across diverse genres and multi-task evaluations, revealing that current MLLMs struggle significantly and degrade sharply with longer durations.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"The chosen video lengths, genres, and tasks in MLVU sufficiently represent the core challenges of real-world long video understanding and that performance on these tasks generalizes beyond the benchmark.","pith_extraction_headline":"MLVU benchmark shows current multimodal models struggle with most long video tasks and degrade sharply on longer clips."},"references":{"count":63,"sample":[{"doi":"","year":2023,"title":"GPT-4 Technical Report","work_id":"b928e041-6991-4c08-8c81-0359e4097c7b","ref_index":1,"cited_arxiv_id":"2303.08774","is_internal_anchor":true},{"doi":"","year":2024,"title":"Anthropic. Claude 3. https://www.anthropic.com/ news/claude-3-family, 2024. 7, 2","work_id":"e9ea5ac3-366b-4128-9649-e6c18dfcadba","ref_index":2,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2024,"title":"Minigpt4-video: Advancing multimodal llms for video understanding with interleaved visual-textual tokens","work_id":"cc937528-86d1-430f-bb5d-4980dbaadd72","ref_index":3,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":null,"title":"Qwen Technical Report","work_id":"bb1fd52f-6b2f-437c-9516-37bdf6eb9be8","ref_index":4,"cited_arxiv_id":"2309.16609","is_internal_anchor":true},{"doi":"","year":2021,"title":"Frozen in time: A joint video and image encoder for end-to- end retrieval","work_id":"dbc0a634-c5d3-4207-8589-9084f58b919d","ref_index":5,"cited_arxiv_id":"","is_internal_anchor":false}],"resolved_work":63,"snapshot_sha256":"d9f5815a74603805b2696b4fe06a39032a108ead5b9be9d2cf5b4582c7eab1c0","internal_anchors":24},"formal_canon":{"evidence_count":2,"snapshot_sha256":"6625c7bfe019846c8f94968c384a39134d4c918e3f265a4782f40ab3d18a8b1d"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":"49f26cf4-8e6d-424a-beee-af7ca6b1887f"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-17T23:39:21Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"YO9rYfBpUiFMhOFP++M2pW+Gsub6bBaL3w5iToOUhlRQH1A+DO/C/exW9Z7vvSQoZH/N50qcYDnnSKQ619VLAA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-07-26T04:48:06.974922Z"},"content_sha256":"558bc9bff1295713c1017588de47b88c4ee9dfcbab61a7743debef18429162b6","schema_version":"1.0","event_id":"sha256:558bc9bff1295713c1017588de47b88c4ee9dfcbab61a7743debef18429162b6"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/T7EF4SUOFTLLSQDQNLC2BWIURX/bundle.json","state_url":"https://pith.science/pith/T7EF4SUOFTLLSQDQNLC2BWIURX/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/T7EF4SUOFTLLSQDQNLC2BWIURX/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-07-26T04:48:06Z","links":{"resolver":"https://pith.science/pith/T7EF4SUOFTLLSQDQNLC2BWIURX","bundle":"https://pith.science/pith/T7EF4SUOFTLLSQDQNLC2BWIURX/bundle.json","state":"https://pith.science/pith/T7EF4SUOFTLLSQDQNLC2BWIURX/state.json","well_known_bundle":"https://pith.science/.well-known/pith/T7EF4SUOFTLLSQDQNLC2BWIURX/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:T7EF4SUOFTLLSQDQNLC2BWIURX","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"41148e2d78f4d31b4e397c1a844d4b6f5ce24671e5f8064a2655bd442342726a","cross_cats_sorted":["cs.AI","cs.CL"],"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-06-06T17:09:32Z","title_canon_sha256":"18a5bf14850be2ac26f321769b1dc07c5948b5228a893e1a555f8f75b88e6ccd"},"schema_version":"1.0","source":{"id":"2406.04264","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2406.04264","created_at":"2026-05-17T23:39:21Z"},{"alias_kind":"arxiv_version","alias_value":"2406.04264v3","created_at":"2026-05-17T23:39:21Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.04264","created_at":"2026-05-17T23:39:21Z"},{"alias_kind":"pith_short_12","alias_value":"T7EF4SUOFTLL","created_at":"2026-05-18T12:33:37Z"},{"alias_kind":"pith_short_16","alias_value":"T7EF4SUOFTLLSQDQ","created_at":"2026-05-18T12:33:37Z"},{"alias_kind":"pith_short_8","alias_value":"T7EF4SUO","created_at":"2026-05-18T12:33:37Z"}],"graph_snapshots":[{"event_id":"sha256:558bc9bff1295713c1017588de47b88c4ee9dfcbab61a7743debef18429162b6","target":"graph","created_at":"2026-05-17T23:39:21Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"The empirical study with 23 latest MLLMs reveals significant room for improvement in today's technique, as all existing methods struggle with most of the evaluation tasks and exhibit severe performance degradation when handling longer videos."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"The chosen video lengths, genres, and tasks in MLVU sufficiently represent the core challenges of real-world long video understanding and that performance on these tasks generalizes beyond the benchmark."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"MLVU is a new benchmark for long video understanding that uses extended videos across diverse genres and multi-task evaluations, revealing that current MLLMs struggle significantly and degrade sharply with longer durations."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"MLVU benchmark shows current multimodal models struggle with most long video tasks and degrade sharply on longer clips."}],"snapshot_sha256":"c886ba5b9cf2cbdfb5ba0c78b921da150b529af8bb67e4b124e6a3446d2abb83"},"formal_canon":{"evidence_count":2,"snapshot_sha256":"6625c7bfe019846c8f94968c384a39134d4c918e3f265a4782f40ab3d18a8b1d"},"paper":{"abstract_excerpt":"The evaluation of Long Video Understanding (LVU) performance poses an important but challenging research problem. Despite previous efforts, the existing video understanding benchmarks are severely constrained by several issues, especially the insufficient lengths of videos, a lack of diversity in video types and evaluation tasks, and the inappropriateness for evaluating LVU performances. To address the above problems, we propose a new benchmark called MLVU (Multi-task Long Video Understanding Benchmark) for the comprehensive and in-depth evaluation of LVU. MLVU presents the following critical ","authors_text":"Boya Wu, Bo Zhang, Bo Zhao, Junjie Zhou, Minghao Qin, Shitao Xiao, Tiejun Huang, Xi Yang, Yan Shu, Yongping Xiong, Zheng Liu, Zhengyang Liang","cross_cats":["cs.AI","cs.CL"],"headline":"MLVU benchmark shows current multimodal models struggle with most long video tasks and degrade sharply on longer clips.","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-06-06T17:09:32Z","title":"MLVU: Benchmarking Multi-task Long Video Understanding"},"references":{"count":63,"internal_anchors":24,"resolved_work":63,"sample":[{"cited_arxiv_id":"2303.08774","doi":"","is_internal_anchor":true,"ref_index":1,"title":"GPT-4 Technical Report","work_id":"b928e041-6991-4c08-8c81-0359e4097c7b","year":2023},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":2,"title":"Anthropic. Claude 3. https://www.anthropic.com/ news/claude-3-family, 2024. 7, 2","work_id":"e9ea5ac3-366b-4128-9649-e6c18dfcadba","year":2024},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":3,"title":"Minigpt4-video: Advancing multimodal llms for video understanding with interleaved visual-textual tokens","work_id":"cc937528-86d1-430f-bb5d-4980dbaadd72","year":2024},{"cited_arxiv_id":"2309.16609","doi":"","is_internal_anchor":true,"ref_index":4,"title":"Qwen Technical Report","work_id":"bb1fd52f-6b2f-437c-9516-37bdf6eb9be8","year":null},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":5,"title":"Frozen in time: A joint video and image encoder for end-to- end retrieval","work_id":"dbc0a634-c5d3-4207-8589-9084f58b919d","year":2021}],"snapshot_sha256":"d9f5815a74603805b2696b4fe06a39032a108ead5b9be9d2cf5b4582c7eab1c0"},"source":{"id":"2406.04264","kind":"arxiv","version":3},"verdict":{"created_at":"2026-05-14T19:49:55.465952Z","id":"49f26cf4-8e6d-424a-beee-af7ca6b1887f","model_set":{"reader":"grok-4.3"},"one_line_summary":"MLVU is a new benchmark for long video understanding that uses extended videos across diverse genres and multi-task evaluations, revealing that current MLLMs struggle significantly and degrade sharply with longer durations.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"MLVU benchmark shows current multimodal models struggle with most long video tasks and degrade sharply on longer clips.","strongest_claim":"The empirical study with 23 latest MLLMs reveals significant room for improvement in today's technique, as all existing methods struggle with most of the evaluation tasks and exhibit severe performance degradation when handling longer videos.","weakest_assumption":"The chosen video lengths, genres, and tasks in MLVU sufficiently represent the core challenges of real-world long video understanding and that performance on these tasks generalizes beyond the benchmark."}},"verdict_id":"49f26cf4-8e6d-424a-beee-af7ca6b1887f"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:1feb9bf7005aa08e00307218042e2ea0c81e7c9bc18f713ff9c838ba500e0ca9","target":"record","created_at":"2026-05-17T23:39:21Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"41148e2d78f4d31b4e397c1a844d4b6f5ce24671e5f8064a2655bd442342726a","cross_cats_sorted":["cs.AI","cs.CL"],"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-06-06T17:09:32Z","title_canon_sha256":"18a5bf14850be2ac26f321769b1dc07c5948b5228a893e1a555f8f75b88e6ccd"},"schema_version":"1.0","source":{"id":"2406.04264","kind":"arxiv","version":3}},"canonical_sha256":"9fc85e4a8e2cd6b940706ac5a0d9148de724540951e474f6b45f20ab24059c3a","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"9fc85e4a8e2cd6b940706ac5a0d9148de724540951e474f6b45f20ab24059c3a","first_computed_at":"2026-05-17T23:39:21.958186Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-05-17T23:39:21.958186Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"KTb9S0ZdoT5X7XkVuO1YvtcJ6lef3HAFS3b7z6E2zvRhYDsraWyyuyAtIBuVa27fzEb/GiJfZqTZnpQUzBx6BA==","signature_status":"signed_v1","signed_at":"2026-05-17T23:39:21.958888Z","signed_message":"canonical_sha256_bytes"},"source_id":"2406.04264","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:1feb9bf7005aa08e00307218042e2ea0c81e7c9bc18f713ff9c838ba500e0ca9","sha256:558bc9bff1295713c1017588de47b88c4ee9dfcbab61a7743debef18429162b6"],"state_sha256":"a7372df17512bd5d8d40827b058d6f8485062879d26bdc8a22f117cedc6d0574"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"SU6nnhvRWur17iV+kAO85s7vbtRn++RjPWGW8+419w32kMV4MFWQ3bIWu1iWQqAikVmaKz0gJ8ExiQXF1nh8CQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-07-26T04:48:06.977368Z","bundle_sha256":"4098f878bd34ea1927ccff85b48fa39b544573462ac06330e34fd3c11601e1e2"}}