{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:2DQVLJV2OC5T5B33LZ4NKRNHK3","short_pith_number":"pith:2DQVLJV2","schema_version":"1.0","canonical_sha256":"d0e155a6ba70bb3e877b5e78d545a756fc3c31b17b53f4b8b6a3e6fb3589e97f","source":{"kind":"arxiv","id":"2412.08646","version":2},"attestation_state":"computed","paper":{"title":"StreamChat: Chatting with Streaming Video","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hongsheng Li, Jan Kautz, Jihao Liu, Jose M. Alvare, Rongyao Fang, Shihao Wang, Shiyi Lan, Zhiding Yu","submitted_at":"2024-12-11T18:59:54Z","abstract_excerpt":"This paper presents StreamChat, a novel approach that enhances the interaction capabilities of Large Multimodal Models (LMMs) with streaming video content. In streaming interaction scenarios, existing methods rely solely on visual information available at the moment a question is posed, resulting in significant delays as the model remains unaware of subsequent changes in the streaming video. StreamChat addresses this limitation by innovatively updating the visual context at each decoding step, ensuring that the model utilizes up-to-date video content throughout the decoding process. Additional"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.08646","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-12-11T18:59:54Z","cross_cats_sorted":[],"title_canon_sha256":"6e3033455ea05fc696a76a5cd597c854d1fa7a445ed65138e5bda928e56cebc0","abstract_canon_sha256":"2552049525232d825cd58744e7da22811a7f52cc0c51d187faf0c4c7426f789a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:41:18.797159Z","signature_b64":"8LgppFsyefIlR4nShBLB2xbTgX7LirGRxysRgnacdD6i4p9p3HYXocrtpgTWutO2Cjp9A3stMO3wwnqBD3VrCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d0e155a6ba70bb3e877b5e78d545a756fc3c31b17b53f4b8b6a3e6fb3589e97f","last_reissued_at":"2026-07-05T10:41:18.796667Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:41:18.796667Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"StreamChat: Chatting with Streaming Video","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hongsheng Li, Jan Kautz, Jihao Liu, Jose M. Alvare, Rongyao Fang, Shihao Wang, Shiyi Lan, Zhiding Yu","submitted_at":"2024-12-11T18:59:54Z","abstract_excerpt":"This paper presents StreamChat, a novel approach that enhances the interaction capabilities of Large Multimodal Models (LMMs) with streaming video content. In streaming interaction scenarios, existing methods rely solely on visual information available at the moment a question is posed, resulting in significant delays as the model remains unaware of subsequent changes in the streaming video. StreamChat addresses this limitation by innovatively updating the visual context at each decoding step, ensuring that the model utilizes up-to-date video content throughout the decoding process. Additional"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.08646","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.08646/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.08646","created_at":"2026-07-05T10:41:18.796725+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.08646v2","created_at":"2026-07-05T10:41:18.796725+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.08646","created_at":"2026-07-05T10:41:18.796725+00:00"},{"alias_kind":"pith_short_12","alias_value":"2DQVLJV2OC5T","created_at":"2026-07-05T10:41:18.796725+00:00"},{"alias_kind":"pith_short_16","alias_value":"2DQVLJV2OC5T5B33","created_at":"2026-07-05T10:41:18.796725+00:00"},{"alias_kind":"pith_short_8","alias_value":"2DQVLJV2","created_at":"2026-07-05T10:41:18.796725+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19849","citing_title":"ViCoStream: Streaming VideoLLMs Can Run Beyond 100 FPS with Stage-Wise Coordinated Inference","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06891","citing_title":"Stream3D-VLM: Online 3D Spatial Understanding with Incremental Geometry Priors","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25621","citing_title":"StreamOV: Streaming Omni-Video Understanding via Evidence-Guided Memory and Response Triggering","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00523","citing_title":"ProactiveLLM: Learning Active Interaction for Streaming Large Language Models","ref_index":86,"is_internal_anchor":false},{"citing_arxiv_id":"2511.21998","citing_title":"Can Multi-Modal LLMs Provide Live Step-by-Step Task Guidance?","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10199","citing_title":"How Should LLMs Listen While Speaking? A Study of User-Stream Routing in Full-Duplex Spoken Dialogue","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19536","citing_title":"LiveVLN: Breaking the Stop-and-Go Loop in Vision-Language Navigation","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06036","citing_title":"CodecSight: Leveraging Video Codec Signals for Efficient Streaming VLM Inference","ref_index":41,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2DQVLJV2OC5T5B33LZ4NKRNHK3","json":"https://pith.science/pith/2DQVLJV2OC5T5B33LZ4NKRNHK3.json","graph_json":"https://pith.science/api/pith-number/2DQVLJV2OC5T5B33LZ4NKRNHK3/graph.json","events_json":"https://pith.science/api/pith-number/2DQVLJV2OC5T5B33LZ4NKRNHK3/events.json","paper":"https://pith.science/paper/2DQVLJV2"},"agent_actions":{"view_html":"https://pith.science/pith/2DQVLJV2OC5T5B33LZ4NKRNHK3","download_json":"https://pith.science/pith/2DQVLJV2OC5T5B33LZ4NKRNHK3.json","view_paper":"https://pith.science/paper/2DQVLJV2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.08646&json=true","fetch_graph":"https://pith.science/api/pith-number/2DQVLJV2OC5T5B33LZ4NKRNHK3/graph.json","fetch_events":"https://pith.science/api/pith-number/2DQVLJV2OC5T5B33LZ4NKRNHK3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2DQVLJV2OC5T5B33LZ4NKRNHK3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2DQVLJV2OC5T5B33LZ4NKRNHK3/action/storage_attestation","attest_author":"https://pith.science/pith/2DQVLJV2OC5T5B33LZ4NKRNHK3/action/author_attestation","sign_citation":"https://pith.science/pith/2DQVLJV2OC5T5B33LZ4NKRNHK3/action/citation_signature","submit_replication":"https://pith.science/pith/2DQVLJV2OC5T5B33LZ4NKRNHK3/action/replication_record"}},"created_at":"2026-07-05T10:41:18.796725+00:00","updated_at":"2026-07-05T10:41:18.796725+00:00"}