{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:MEDIXOOBGFBFHKJMZQCRDFK33D","short_pith_number":"pith:MEDIXOOB","schema_version":"1.0","canonical_sha256":"61068bb9c1314253a92ccc0511955bd8c4fb5061740bb46d67aa3765f6e27bd0","source":{"kind":"arxiv","id":"2409.20018","version":2},"attestation_state":"computed","paper":{"title":"Visual Context Window Extension: A New Perspective for Long Video Understanding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hongchen Wei, Zhenzhong Chen","submitted_at":"2024-09-30T07:25:16Z","abstract_excerpt":"Large Multimodal Models (LMMs) have demonstrated impressive performance in short video understanding tasks but face great challenges when applied to long video understanding. In contrast, Large Language Models (LLMs) exhibit outstanding capabilities in modeling long texts. Existing work attempts to address this issue by introducing long video-text pairs during training. However, these approaches require substantial computational and data resources. In this paper, we tackle the challenge of long video understanding from the perspective of context windows, aiming to apply LMMs to long video task"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.20018","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-09-30T07:25:16Z","cross_cats_sorted":[],"title_canon_sha256":"94278d74d6da1f62cd091c981a977ea899e16075cc1eb75ce4f5879a561d2131","abstract_canon_sha256":"04963343b4c960388a9cdcc59ea6186d9a7d4794f2e3d25efee5661b30972b9c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:14:39.841557Z","signature_b64":"n+rdA9HT7ZWtaR8036zbRYquGYrn/Vw6BTlHviz+BGarl0ZG7bFFQ54dOe5P6tnVbEFdN27XGF1WRr0h9tWiCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"61068bb9c1314253a92ccc0511955bd8c4fb5061740bb46d67aa3765f6e27bd0","last_reissued_at":"2026-07-05T09:14:39.841104Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:14:39.841104Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Visual Context Window Extension: A New Perspective for Long Video Understanding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hongchen Wei, Zhenzhong Chen","submitted_at":"2024-09-30T07:25:16Z","abstract_excerpt":"Large Multimodal Models (LMMs) have demonstrated impressive performance in short video understanding tasks but face great challenges when applied to long video understanding. In contrast, Large Language Models (LLMs) exhibit outstanding capabilities in modeling long texts. Existing work attempts to address this issue by introducing long video-text pairs during training. However, these approaches require substantial computational and data resources. In this paper, we tackle the challenge of long video understanding from the perspective of context windows, aiming to apply LMMs to long video task"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.20018","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.20018/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.20018","created_at":"2026-07-05T09:14:39.841168+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.20018v2","created_at":"2026-07-05T09:14:39.841168+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.20018","created_at":"2026-07-05T09:14:39.841168+00:00"},{"alias_kind":"pith_short_12","alias_value":"MEDIXOOBGFBF","created_at":"2026-07-05T09:14:39.841168+00:00"},{"alias_kind":"pith_short_16","alias_value":"MEDIXOOBGFBFHKJM","created_at":"2026-07-05T09:14:39.841168+00:00"},{"alias_kind":"pith_short_8","alias_value":"MEDIXOOB","created_at":"2026-07-05T09:14:39.841168+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12195","citing_title":"InternVideo3: Agentify Foundation Models with Multimodal Contextual Reasoning","ref_index":284,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11913","citing_title":"From Content to Knowledge: Lightning Fast Long-Video Understanding with Neural Knowledge Representations","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05748","citing_title":"UNIVID: Unified Vision-Language Model for Video Moderation","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2501.00574","citing_title":"VideoChat-Flash: Hierarchical Compression for Long-Context Video Modeling","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06185","citing_title":"Event-Causal RAG: A Retrieval-Augmented Generation Framework for Long Video Reasoning in Complex Scenarios","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MEDIXOOBGFBFHKJMZQCRDFK33D","json":"https://pith.science/pith/MEDIXOOBGFBFHKJMZQCRDFK33D.json","graph_json":"https://pith.science/api/pith-number/MEDIXOOBGFBFHKJMZQCRDFK33D/graph.json","events_json":"https://pith.science/api/pith-number/MEDIXOOBGFBFHKJMZQCRDFK33D/events.json","paper":"https://pith.science/paper/MEDIXOOB"},"agent_actions":{"view_html":"https://pith.science/pith/MEDIXOOBGFBFHKJMZQCRDFK33D","download_json":"https://pith.science/pith/MEDIXOOBGFBFHKJMZQCRDFK33D.json","view_paper":"https://pith.science/paper/MEDIXOOB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.20018&json=true","fetch_graph":"https://pith.science/api/pith-number/MEDIXOOBGFBFHKJMZQCRDFK33D/graph.json","fetch_events":"https://pith.science/api/pith-number/MEDIXOOBGFBFHKJMZQCRDFK33D/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MEDIXOOBGFBFHKJMZQCRDFK33D/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MEDIXOOBGFBFHKJMZQCRDFK33D/action/storage_attestation","attest_author":"https://pith.science/pith/MEDIXOOBGFBFHKJMZQCRDFK33D/action/author_attestation","sign_citation":"https://pith.science/pith/MEDIXOOBGFBFHKJMZQCRDFK33D/action/citation_signature","submit_replication":"https://pith.science/pith/MEDIXOOBGFBFHKJMZQCRDFK33D/action/replication_record"}},"created_at":"2026-07-05T09:14:39.841168+00:00","updated_at":"2026-07-05T09:14:39.841168+00:00"}