{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:DIQG5CLEMSLR3ZUW6HK3CND5S7","short_pith_number":"pith:DIQG5CLE","schema_version":"1.0","canonical_sha256":"1a206e896464971de696f1d5b1347d97df01976fb1036c6b8321f3cf4b14002a","source":{"kind":"arxiv","id":"2411.03628","version":1},"attestation_state":"computed","paper":{"title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Chi Chen, Fuwen Luo, Junming Lin, Maosong Sun, Peng Li, Yang Liu, Zheng Fang, Zihao Wan","submitted_at":"2024-11-06T02:50:30Z","abstract_excerpt":"The rapid development of Multimodal Large Language Models (MLLMs) has expanded their capabilities from image comprehension to video understanding. However, most of these MLLMs focus primarily on offline video comprehension, necessitating extensive processing of all video frames before any queries can be made. This presents a significant gap compared to the human ability to watch, listen, think, and respond to streaming inputs in real time, highlighting the limitations of current MLLMs. In this paper, we introduce StreamingBench, the first comprehensive benchmark designed to evaluate the stream"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.03628","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-11-06T02:50:30Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"f5c7fa5c9bbdcd9191864ffc46532cd44960d2c531b45cd310138d8c568718cc","abstract_canon_sha256":"8d8e17727a8fee5ecd5baeebaa5d16c339f55ef9be8b7a136736051cd63f049d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:31:53.323231Z","signature_b64":"ESLMn3TwxPZriKK0RvfCQDvb8RugXGB1Go35saInNa37lOWvCtn74Gj3SnrvEeI+x7/Dil1tUihS90NZJrsPCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1a206e896464971de696f1d5b1347d97df01976fb1036c6b8321f3cf4b14002a","last_reissued_at":"2026-07-05T09:31:53.322725Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:31:53.322725Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Chi Chen, Fuwen Luo, Junming Lin, Maosong Sun, Peng Li, Yang Liu, Zheng Fang, Zihao Wan","submitted_at":"2024-11-06T02:50:30Z","abstract_excerpt":"The rapid development of Multimodal Large Language Models (MLLMs) has expanded their capabilities from image comprehension to video understanding. However, most of these MLLMs focus primarily on offline video comprehension, necessitating extensive processing of all video frames before any queries can be made. This presents a significant gap compared to the human ability to watch, listen, think, and respond to streaming inputs in real time, highlighting the limitations of current MLLMs. In this paper, we introduce StreamingBench, the first comprehensive benchmark designed to evaluate the stream"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.03628","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.03628/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.03628","created_at":"2026-07-05T09:31:53.322788+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.03628v1","created_at":"2026-07-05T09:31:53.322788+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.03628","created_at":"2026-07-05T09:31:53.322788+00:00"},{"alias_kind":"pith_short_12","alias_value":"DIQG5CLEMSLR","created_at":"2026-07-05T09:31:53.322788+00:00"},{"alias_kind":"pith_short_16","alias_value":"DIQG5CLEMSLR3ZUW","created_at":"2026-07-05T09:31:53.322788+00:00"},{"alias_kind":"pith_short_8","alias_value":"DIQG5CLE","created_at":"2026-07-05T09:31:53.322788+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":29,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24422","citing_title":"EgoSAT: A Comprehensive Benchmark of Egocentric Streaming Interaction Understanding","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26762","citing_title":"ProtoKV: Streaming Video Understanding under Delayed Query with Summary-State Memory","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17360","citing_title":"Omni-DuplexEval: Evaluating Real-time Duplex Omni-modal Interaction","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01751","citing_title":"MedStreamBench: A Time-Aware Benchmark for Streaming and Proactive Medical Video Understanding","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09547","citing_title":"Streaming Interventions: Can Video Large Language Models Correct Mistakes as They Occur?","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06991","citing_title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","ref_index":101,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02482","citing_title":"X-Stream: Exploring MLLMs as Multiplexers for Multi-Stream Understanding","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07639","citing_title":"MOSS-Video-Preview: Toward Real-Time Video Understanding via Cross-Attention","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02482","citing_title":"X-Stream: Exploring MLLMs as Multiplexers for Multi-Stream Understanding","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25621","citing_title":"StreamOV: Streaming Omni-Video Understanding via Evidence-Guided Memory and Response Triggering","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31557","citing_title":"EGOSTREAM: A Diagnostic Benchmark for Streaming Episodic Memory in Egocentric Vision","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22983","citing_title":"LiveServe: Interaction-Aware Serving for Real-Time Omni-Modal LLMs","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2505.15269","citing_title":"LiveVLM: Efficient Online Video Understanding via Streaming-Oriented KV Cache and Retrieval","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22269","citing_title":"MuKV: Multi-Grained KV Cache Compression for Long Streaming Video Question-Answering","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17360","citing_title":"Omni-DuplexEval: Evaluating Real-time Duplex Omni-modal Interaction","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17921","citing_title":"An Efficient Streaming Video Understanding Framework with Agentic Control","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2502.04326","citing_title":"WorldSense: Evaluating Real-world Omnimodal Understanding for Multimodal LLMs","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2512.01707","citing_title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2603.20633","citing_title":"Seed1.8 Model Card: Towards Generalized Real-World Agency","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07575","citing_title":"Response-G1: Explicit Scene Graph Modeling for Proactive Streaming Video Understanding","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24317","citing_title":"Don't Pause! Every prediction matters in a streaming video","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19536","citing_title":"LiveVLN: Breaking the Stop-and-Go Loop in Vision-Language Navigation","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01858","citing_title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11411","citing_title":"Online Reasoning Video Object Segmentation","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07634","citing_title":"VSAS-Bench: Real-Time Evaluation of Visual Streaming Assistant Models","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DIQG5CLEMSLR3ZUW6HK3CND5S7","json":"https://pith.science/pith/DIQG5CLEMSLR3ZUW6HK3CND5S7.json","graph_json":"https://pith.science/api/pith-number/DIQG5CLEMSLR3ZUW6HK3CND5S7/graph.json","events_json":"https://pith.science/api/pith-number/DIQG5CLEMSLR3ZUW6HK3CND5S7/events.json","paper":"https://pith.science/paper/DIQG5CLE"},"agent_actions":{"view_html":"https://pith.science/pith/DIQG5CLEMSLR3ZUW6HK3CND5S7","download_json":"https://pith.science/pith/DIQG5CLEMSLR3ZUW6HK3CND5S7.json","view_paper":"https://pith.science/paper/DIQG5CLE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.03628&json=true","fetch_graph":"https://pith.science/api/pith-number/DIQG5CLEMSLR3ZUW6HK3CND5S7/graph.json","fetch_events":"https://pith.science/api/pith-number/DIQG5CLEMSLR3ZUW6HK3CND5S7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DIQG5CLEMSLR3ZUW6HK3CND5S7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DIQG5CLEMSLR3ZUW6HK3CND5S7/action/storage_attestation","attest_author":"https://pith.science/pith/DIQG5CLEMSLR3ZUW6HK3CND5S7/action/author_attestation","sign_citation":"https://pith.science/pith/DIQG5CLEMSLR3ZUW6HK3CND5S7/action/citation_signature","submit_replication":"https://pith.science/pith/DIQG5CLEMSLR3ZUW6HK3CND5S7/action/replication_record"}},"created_at":"2026-07-05T09:31:53.322788+00:00","updated_at":"2026-07-05T09:31:53.322788+00:00"}