{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:PBPOO6MDUQPWI5S7BUJOTVTE5I","short_pith_number":"pith:PBPOO6MD","schema_version":"1.0","canonical_sha256":"785ee77983a41f64765f0d12e9d664ea178d74dd281ad1b243853da266439861","source":{"kind":"arxiv","id":"2405.16009","version":1},"attestation_state":"computed","paper":{"title":"Streaming Long Video Understanding with Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dahua Lin, Jiaqi Wang, Pan Zhang, Rui Qian, Shuangrui Ding, Xiaoyi Dong, Yuhang Zang","submitted_at":"2024-05-25T02:22:09Z","abstract_excerpt":"This paper presents VideoStreaming, an advanced vision-language large model (VLLM) for video understanding, that capably understands arbitrary-length video with a constant number of video tokens streamingly encoded and adaptively selected. The challenge of video understanding in the vision language area mainly lies in the significant computational burden caused by the great number of tokens extracted from long videos. Previous works rely on sparse sampling or frame compression to reduce tokens. However, such approaches either disregard temporal information in a long time span or sacrifice spat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.16009","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-05-25T02:22:09Z","cross_cats_sorted":[],"title_canon_sha256":"0a8113d302cb279e806435d879afa00512142b4f3f57c40cce27e41946f188cb","abstract_canon_sha256":"6933b3f90328fde2142989fe32b5edac908f320875793d4560350ec34183b8d6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:23:00.441324Z","signature_b64":"EeG6jycazlmQ8Ual008bUEVvKkRH9MRpbiQHT2ioDP2rBolj1xAMdu2ENFY51VxYWQ7w0Q9GKRRp84l8J8UeAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"785ee77983a41f64765f0d12e9d664ea178d74dd281ad1b243853da266439861","last_reissued_at":"2026-07-05T08:23:00.440850Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:23:00.440850Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Streaming Long Video Understanding with Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dahua Lin, Jiaqi Wang, Pan Zhang, Rui Qian, Shuangrui Ding, Xiaoyi Dong, Yuhang Zang","submitted_at":"2024-05-25T02:22:09Z","abstract_excerpt":"This paper presents VideoStreaming, an advanced vision-language large model (VLLM) for video understanding, that capably understands arbitrary-length video with a constant number of video tokens streamingly encoded and adaptively selected. The challenge of video understanding in the vision language area mainly lies in the significant computational burden caused by the great number of tokens extracted from long videos. Previous works rely on sparse sampling or frame compression to reduce tokens. However, such approaches either disregard temporal information in a long time span or sacrifice spat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.16009","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.16009/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.16009","created_at":"2026-07-05T08:23:00.440918+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.16009v1","created_at":"2026-07-05T08:23:00.440918+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.16009","created_at":"2026-07-05T08:23:00.440918+00:00"},{"alias_kind":"pith_short_12","alias_value":"PBPOO6MDUQPW","created_at":"2026-07-05T08:23:00.440918+00:00"},{"alias_kind":"pith_short_16","alias_value":"PBPOO6MDUQPWI5S7","created_at":"2026-07-05T08:23:00.440918+00:00"},{"alias_kind":"pith_short_8","alias_value":"PBPOO6MD","created_at":"2026-07-05T08:23:00.440918+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12195","citing_title":"InternVideo3: Agentify Foundation Models with Multimodal Contextual Reasoning","ref_index":169,"is_internal_anchor":false},{"citing_arxiv_id":"2505.23747","citing_title":"Spatial-MLLM: Boosting MLLM Capabilities in Visual-based Spatial Intelligence","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2510.09608","citing_title":"StreamingVLM: Real-Time Understanding for Infinite Video Streams","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2505.23747","citing_title":"Spatial-MLLM: Boosting MLLM Capabilities in Visual-based Spatial Intelligence","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14310","citing_title":"CoRDS: Coreset-based Representative and Diverse Selection for Streaming Video Understanding","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2501.13106","citing_title":"VideoLLaMA 3: Frontier Multimodal Foundation Models for Image and Video Understanding","ref_index":169,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PBPOO6MDUQPWI5S7BUJOTVTE5I","json":"https://pith.science/pith/PBPOO6MDUQPWI5S7BUJOTVTE5I.json","graph_json":"https://pith.science/api/pith-number/PBPOO6MDUQPWI5S7BUJOTVTE5I/graph.json","events_json":"https://pith.science/api/pith-number/PBPOO6MDUQPWI5S7BUJOTVTE5I/events.json","paper":"https://pith.science/paper/PBPOO6MD"},"agent_actions":{"view_html":"https://pith.science/pith/PBPOO6MDUQPWI5S7BUJOTVTE5I","download_json":"https://pith.science/pith/PBPOO6MDUQPWI5S7BUJOTVTE5I.json","view_paper":"https://pith.science/paper/PBPOO6MD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.16009&json=true","fetch_graph":"https://pith.science/api/pith-number/PBPOO6MDUQPWI5S7BUJOTVTE5I/graph.json","fetch_events":"https://pith.science/api/pith-number/PBPOO6MDUQPWI5S7BUJOTVTE5I/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PBPOO6MDUQPWI5S7BUJOTVTE5I/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PBPOO6MDUQPWI5S7BUJOTVTE5I/action/storage_attestation","attest_author":"https://pith.science/pith/PBPOO6MDUQPWI5S7BUJOTVTE5I/action/author_attestation","sign_citation":"https://pith.science/pith/PBPOO6MDUQPWI5S7BUJOTVTE5I/action/citation_signature","submit_replication":"https://pith.science/pith/PBPOO6MDUQPWI5S7BUJOTVTE5I/action/replication_record"}},"created_at":"2026-07-05T08:23:00.440918+00:00","updated_at":"2026-07-05T08:23:00.440918+00:00"}