{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:LZ7G55V5BQRFPSOUSYX2JMAQQB","short_pith_number":"pith:LZ7G55V5","schema_version":"1.0","canonical_sha256":"5e7e6ef6bd0c2257c9d4962fa4b0108062854dc7c2743e24b3777311c915d0a8","source":{"kind":"arxiv","id":"2409.14485","version":4},"attestation_state":"computed","paper":{"title":"Video-XL: Extra-Long Vision Language Model for Hour-Scale Video Understanding","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bo Zhao, Junjie Zhou, Minghao Qin, Peitian Zhang, Tiejun Huang, Yan Shu, Zheng Liu, Zhengyang Liang","submitted_at":"2024-09-22T15:13:31Z","abstract_excerpt":"Long video understanding poses a significant challenge for current Multi-modal Large Language Models (MLLMs). Notably, the MLLMs are constrained by their limited context lengths and the substantial costs while processing long videos. Although several existing methods attempt to reduce visual tokens, their strategies encounter severe bottleneck, restricting MLLMs' ability to perceive fine-grained visual details. In this work, we propose Video-XL, a novel approach that leverages MLLMs' inherent key-value (KV) sparsification capacity to condense the visual input. Specifically, we introduce a new "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.14485","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-09-22T15:13:31Z","cross_cats_sorted":[],"title_canon_sha256":"79f441616e1a6aaca7d86211f691fd3b19dcfd3482f36774f3642f33b9dd31ae","abstract_canon_sha256":"6ba1e3e75a4028f5ba6a9af7c9ed26c1a8990d0b1ab273c244e881c982f861c4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:46:51.996908Z","signature_b64":"64CMhWUGxi2F0zEna/QqeCIN7b5CqtDsiHfDApRFWMoYOJS+pzBZvACDgnucnsOC0dpz8RZHf/3RTDnoA2snAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5e7e6ef6bd0c2257c9d4962fa4b0108062854dc7c2743e24b3777311c915d0a8","last_reissued_at":"2026-07-05T09:46:51.996382Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:46:51.996382Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Video-XL: Extra-Long Vision Language Model for Hour-Scale Video Understanding","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bo Zhao, Junjie Zhou, Minghao Qin, Peitian Zhang, Tiejun Huang, Yan Shu, Zheng Liu, Zhengyang Liang","submitted_at":"2024-09-22T15:13:31Z","abstract_excerpt":"Long video understanding poses a significant challenge for current Multi-modal Large Language Models (MLLMs). Notably, the MLLMs are constrained by their limited context lengths and the substantial costs while processing long videos. Although several existing methods attempt to reduce visual tokens, their strategies encounter severe bottleneck, restricting MLLMs' ability to perceive fine-grained visual details. In this work, we propose Video-XL, a novel approach that leverages MLLMs' inherent key-value (KV) sparsification capacity to condense the visual input. Specifically, we introduce a new "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.14485","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.14485/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.14485","created_at":"2026-07-05T09:46:51.996447+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.14485v4","created_at":"2026-07-05T09:46:51.996447+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.14485","created_at":"2026-07-05T09:46:51.996447+00:00"},{"alias_kind":"pith_short_12","alias_value":"LZ7G55V5BQRF","created_at":"2026-07-05T09:46:51.996447+00:00"},{"alias_kind":"pith_short_16","alias_value":"LZ7G55V5BQRFPSOU","created_at":"2026-07-05T09:46:51.996447+00:00"},{"alias_kind":"pith_short_8","alias_value":"LZ7G55V5","created_at":"2026-07-05T09:46:51.996447+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21734","citing_title":"HPP: Hierarchical Programmatic Probing for Long Video Understanding by Decoupling Perception and Reasoning","ref_index":139,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12195","citing_title":"InternVideo3: Agentify Foundation Models with Multimodal Contextual Reasoning","ref_index":253,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11740","citing_title":"UniReason-Med: A Shared Grounded Reasoning Interface for 2D-to-3D Transfer in Medical VQA","ref_index":111,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06532","citing_title":"GOPAgen: Motion-Aware and Efficient Agentic Long-Video Understanding with Structural Memory and Hierarchical Reasoning","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2412.04468","citing_title":"NVILA: Efficient Frontier Visual Language Models","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2501.00574","citing_title":"VideoChat-Flash: Hierarchical Compression for Long-Context Video Modeling","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2501.12386","citing_title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2603.27259","citing_title":"Seeing the Scene Matters: Revealing Forgetting in Video Understanding Models with a Scene-Aware Long-Video Benchmark","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2406.04264","citing_title":"MLVU: Benchmarking Multi-task Long Video Understanding","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11627","citing_title":"POINTS-Long: Adaptive Dual-Mode Visual Reasoning in MLLMs","ref_index":73,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14149","citing_title":"One Token per Highly Selective Frame: Towards Extreme Compression for Long Video Understanding","ref_index":56,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LZ7G55V5BQRFPSOUSYX2JMAQQB","json":"https://pith.science/pith/LZ7G55V5BQRFPSOUSYX2JMAQQB.json","graph_json":"https://pith.science/api/pith-number/LZ7G55V5BQRFPSOUSYX2JMAQQB/graph.json","events_json":"https://pith.science/api/pith-number/LZ7G55V5BQRFPSOUSYX2JMAQQB/events.json","paper":"https://pith.science/paper/LZ7G55V5"},"agent_actions":{"view_html":"https://pith.science/pith/LZ7G55V5BQRFPSOUSYX2JMAQQB","download_json":"https://pith.science/pith/LZ7G55V5BQRFPSOUSYX2JMAQQB.json","view_paper":"https://pith.science/paper/LZ7G55V5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.14485&json=true","fetch_graph":"https://pith.science/api/pith-number/LZ7G55V5BQRFPSOUSYX2JMAQQB/graph.json","fetch_events":"https://pith.science/api/pith-number/LZ7G55V5BQRFPSOUSYX2JMAQQB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LZ7G55V5BQRFPSOUSYX2JMAQQB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LZ7G55V5BQRFPSOUSYX2JMAQQB/action/storage_attestation","attest_author":"https://pith.science/pith/LZ7G55V5BQRFPSOUSYX2JMAQQB/action/author_attestation","sign_citation":"https://pith.science/pith/LZ7G55V5BQRFPSOUSYX2JMAQQB/action/citation_signature","submit_replication":"https://pith.science/pith/LZ7G55V5BQRFPSOUSYX2JMAQQB/action/replication_record"}},"created_at":"2026-07-05T09:46:51.996447+00:00","updated_at":"2026-07-05T09:46:51.996447+00:00"}