{"as_of":"2026-08-06T22:45:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:6d02a08ed3954700bad787e1c14da74590ccdd42851dc5e864d88b6d7463c6ce","coverage":[{"denominator":120,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-27T22:11:01.690237Z","state":"measured"},{"denominator":101,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":101,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-27T01:12:46.295455Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-03T20:38:56.159073Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"cited_work":{"arxiv_id":"2606.06991","doi":null,"metadata_source":"pith","pith_arxiv_id":"2606.06991","snapshot_observed_at":"2026-07-03T20:38:56.159073Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","venue":"cs.CV","work_id":"b0e2c2e3-5b6e-4821-9097-d95537244496","year":2026},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/2606.06991","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:d281b451ce4468bcc5517d0d61bd8b0c9b63ee5afbf4da5f83ebed7e16bc2937","observation_id":"dd0c6591-e0ee-474c-b682-973138cd64dd","resolution":{"observed_at":"2026-07-03T20:38:56.160538Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2606.06991/citation-record","integrity":"/paper/2606.06991/integrity","json":"/paper/2606.06991/citation-record.json","paper":"/paper/2606.06991"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2404.03413","last_updated":"2024-04-04T12:46:01Z","snapshot_observed_at":"2026-07-06T17:55:36.748672Z","submitted_at":"2024-04-04T12:46:01Z","title":"MiniGPT4-Video: Advancing Multimodal LLMs for Video Understanding with Interleaved Visual-Textual Tokens","version":1},"cited_work":{"arxiv_id":"2404.03413","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.03413","snapshot_observed_at":"2026-07-04T06:39:37.386392Z","title":"Minigpt4-video: Advancing multimodal llms for video understanding with interleaved visual-textual tokens","venue":null,"work_id":"cc937528-86d1-430f-bb5d-4980dbaadd72","year":2024},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"cited_paper":"/paper/2404.03413","citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:48c778a48131b704836b3126a3a259c1703d0e27810b088764a911944f8b7e97","observation_id":"e44ff72d-fe3d-44eb-90f9-764ea821fa25","resolution":{"observed_at":"2026-07-02T17:07:12.876383Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05424","last_updated":"2024-06-10T01:36:53Z","snapshot_observed_at":"2026-07-06T15:40:24.127663Z","submitted_at":"2023-06-08T17:59:56Z","title":"Video-ChatGPT: Towards Detailed Video Understanding via Large Vision and Language Models","version":2},"cited_work":{"arxiv_id":"2306.05424","doi":"10.48550/arxiv.2306.05424","metadata_source":"pith","pith_arxiv_id":"2306.05424","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Video-ChatGPT: Towards Detailed Video Understanding via Large Vision and Language Models","venue":"cs.CV","work_id":"51f627f4-8fae-4882-a3e9-abdf932ef27b","year":2023},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"cited_paper":"/paper/2306.05424","citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:0c24eeab55f2ca1efd4dc319a947d31cb8ce35d12df5095bb63bf98f619e06d2","observation_id":"1a891852-1a2d-428e-816b-60dd59de663f","resolution":{"observed_at":"2026-07-02T17:07:12.864217Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":"2305.06355","doi":"10.48550/arxiv.2305.06355","metadata_source":"pith","pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VideoChat: Chat-Centric Video Understanding","venue":"cs.CV","work_id":"07461eec-156c-4054-a28e-b84bc53bf6e1","year":2023},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:7d51dad474799466e5f29b0e55940248ef11a2704cd684885e26dc152af3174a","observation_id":"51ee5fd0-7ae1-41e2-a6f0-8d1600b3246a","resolution":{"observed_at":"2026-07-02T17:07:12.795455Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Vid2seq: Large-scale pretraining of a visual language model for dense video captioning,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:0f2614b02cfec51db0eca9dccbf4058bfaa78d02b09c16f90729b253f55b8ac9","observation_id":"f217b3a6-fd3e-4fae-8576-c32e802f9c1f","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2212.03191","last_updated":"2022-12-07T12:20:55Z","snapshot_observed_at":"2026-07-06T14:27:34.639236Z","submitted_at":"2022-12-06T18:09:49Z","title":"InternVideo: General Video Foundation Models via Generative and Discriminative Learning","version":2},"cited_work":{"arxiv_id":"2212.03191","doi":"10.48550/arxiv.2212.03191","metadata_source":"pith","pith_arxiv_id":"2212.03191","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo: General Video Foundation Models via Generative and Discriminative Learning","venue":"cs.CV","work_id":"780aaeee-ac26-46b1-b6ff-64a7a624e694","year":2022},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"cited_paper":"/paper/2212.03191","citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:cc257ed0a6e89e9bc51c41b1b6e6dff22e2ba968de898ed4ac6d687374d50dd5","observation_id":"783c631f-84aa-4d63-91a5-94c250fdfd21","resolution":{"observed_at":"2026-07-02T17:07:12.866731Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Cat+: Investigating and enhancing audio-visual understanding in large language models,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:b986776beb17709be132734d1c1e2b4c4cabd0a91cb62ce46076cdb89894556a","observation_id":"f041428e-de8c-40d3-bb42-9bdb76454b86","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Motionllm: Understanding human behaviors from human motions and videos,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:f2a4ab7b2b38b801dec85ab333e3881520384262b2be5c359f638ed325ca6f3c","observation_id":"347ae490-2311-4983-b3b0-10409177c9df","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.07476","last_updated":"2024-10-30T06:49:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-11T17:22:23Z","title":"VideoLLaMA 2: Advancing Spatial-Temporal Modeling and Audio Understanding in Video-LLMs","version":3},"cited_work":{"arxiv_id":"2406.07476","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.07476","snapshot_observed_at":"2026-07-04T16:49:57.279303Z","title":"VideoLLaMA 2: Advancing Spatial-Temporal Modeling and Audio Understanding in Video-LLMs","venue":"cs.CV","work_id":"ccfc3f89-c510-45f1-8a35-ed1a56c0ae5c","year":2024},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"cited_paper":"/paper/2406.07476","citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:1e35fc30b37a0825f9e8d4022b2a7cde6846ece631ebb59bf8037a0ae034c4f1","observation_id":"bca48a04-5b63-4239-8ebb-2a1e2178ad89","resolution":{"observed_at":"2026-07-02T17:07:12.809524Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12961","last_updated":"2025-02-27T06:09:46Z","snapshot_observed_at":"2026-07-06T19:18:17.523057Z","submitted_at":"2024-09-19T17:59:51Z","title":"Oryx MLLM: On-Demand Spatial-Temporal Understanding at Arbitrary Resolution","version":4},"cited_work":{"arxiv_id":"2409.12961","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2409.12961","snapshot_observed_at":"2026-07-04T13:09:50.847114Z","title":"Oryx mllm: On- demand spatial-temporal understanding at arbitrary resolution","venue":null,"work_id":"4a905ca2-7426-40db-a509-3452624d3ae5","year":2024},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"cited_paper":"/paper/2409.12961","citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:abc8e8a5b51632c527dd98c99e525fc57f1ae6319ae124f430643c41d5cc42d1","observation_id":"687ae146-f056-42c1-bd57-d3b9df6a016b","resolution":{"observed_at":"2026-07-02T17:07:12.842034Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Timechat: A time-sensitive multimodal large language model for long video understanding,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:9a7313b45318ac8175a5c9eaf61cbdcf10cc59d08294ce4ceeb6c632a8da800e","observation_id":"56177e63-364b-4896-b1c1-32a93ade46e9","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Hier-egopack: Hierarchical egocentric video understanding with diverse task perspectives,","venue":null,"work_id":null,"year":1917},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:6a3c0d8c77e64d181d1b417b102608f355164b10f75fb62f312596985d74fed2","observation_id":"9c83236c-1277-4bf2-91ae-6d15731c4a4e","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:a0343ffa53418db863fac2fbaf9c122deb8c4b0145355e6f43e85766687546ba","observation_id":"f17c9624-dcc6-4b32-9028-2268977d50fc","resolution":{"observed_at":"2026-07-02T17:07:12.826114Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.17176","last_updated":"2024-04-26T06:17:04Z","snapshot_observed_at":"2026-07-06T18:05:59.538624Z","submitted_at":"2024-04-26T06:17:04Z","title":"MovieChat+: Question-aware Sparse Memory for Long Video Question Answering","version":1},"cited_work":{"arxiv_id":"2404.17176","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.17176","snapshot_observed_at":"2026-07-03T20:38:56.117133Z","title":"Moviechat+: Question-aware sparse memory for long video question answering","venue":null,"work_id":"e6a54789-a1e5-467c-8777-bbb1d564bd4f","year":2024},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"cited_paper":"/paper/2404.17176","citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:0fa26e272efb5feb19a1bfa2dfa45796cad543dc347829708a095696035d31ce","observation_id":"f5be6b5e-1180-4d3a-8fdf-bff484473869","resolution":{"observed_at":"2026-07-02T17:07:12.889758Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Ma-lmm: Memory-augmented large multimodal model for long-term video understanding,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:666c59d580b96aa6bad8d8ec4744cf5c451700dd7ccc0dfba12de789c13f7c9d","observation_id":"d363a032-1ae2-48ab-bfaf-c9fb289d84dc","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2409.02889","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T09:39:46.752891Z","title":"Longllava: Scaling multi-modal llms to 1000 images efficiently via hybrid architecture","venue":null,"work_id":"c93e268d-1703-46de-8123-15c73f06bc0b","year":2024},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:33ea0a49b43a382c6f715bb6f62c37303017b0a4ce588c77f66610079e4304ed","observation_id":"8173de8a-3b4c-4445-84ba-8ab655a6aa64","resolution":{"observed_at":"2026-07-02T17:07:12.855632Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Longvila: Scaling long-context visual language models for long videos,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:434bc144ec6b1d1578c646f45e8438157e134cee8da182994a577abc8c5555bf","observation_id":"81fc4cfe-e144-4d78-b1f4-5742af59720c","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.16852","last_updated":"2024-07-01T02:59:29Z","snapshot_observed_at":"2026-07-06T18:36:12.298104Z","submitted_at":"2024-06-24T17:58:06Z","title":"Long Context Transfer from Language to Vision","version":2},"cited_work":{"arxiv_id":"2406.16852","doi":"10.48550/arxiv.2406.16852","metadata_source":"pith","pith_arxiv_id":"2406.16852","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Long Context Transfer from Language to Vision","venue":"cs.CV","work_id":"52f1b946-568f-4819-9d8a-a87296f8852d","year":2024},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"cited_paper":"/paper/2406.16852","citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:0e771e6e0e277ef503edaf7c97e624e66b84abfe7508c0c18703f075ce53ecd9","observation_id":"e27fd12c-64af-4ba7-88b3-6b57edf869bb","resolution":{"observed_at":"2026-07-02T17:07:12.876722Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Momentor++: Advancing video large language models with fine-grained long video reasoning,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:1e7f4708ff24ad117fd264a888979f850202005db094e8a3b5f41183a3dfb6a5","observation_id":"a8bb0fa2-d1eb-412d-8eff-688c1ea46bc7","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Selongvlm: Empowering long video language models with self-corrective clip selection,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:3d6919c9deefb0a325571d65013ba86723b728128e99f3a3981df32a45da4460","observation_id":"be19c702-97c4-4b6c-8744-afab3935855f","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Ego-r1: Agentic chain-of-tool-thought for ultra- long egocentric video reasoning,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:dfe118c02bc61e87f60fe5dd2e02ec59d793f3e2ca6496429a12dfd8be506ac9","observation_id":"08d4febe-2674-4aed-86e3-12008e935641","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Interaction methods for smart glasses: A survey,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:fee796b9e45ad9ad8f7e78bbf787ea6ed1b30dcc49ba585e1c3ddf9171fea5f7","observation_id":"22a8f6d5-0ef1-4327-95fe-cdab7aaffdd9","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"A head-mounted three dimensional display,","venue":null,"work_id":null,"year":1968},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:96b7eee36175ea0933700797bd53ed3ddf1c298ae422f117752793958d06bbde","observation_id":"3502ee20-345a-4410-84ac-c55a95aaf25f","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Horn,Robot vision","venue":null,"work_id":null,"year":1986},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:0a0f3389daef9d4171e0d886c5220ec8d079b5e5b7e8a80bdb6a1158c55d56dc","observation_id":"1e52514e-baf1-4fd6-a96a-91555d5b7deb","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Videollm-online: Online video large language model for streaming video,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:78304c885849f4c3538d02cc581ba39cd88a571303826414f12b36812fc4cd27","observation_id":"c56ad348-8324-4a9c-b77a-9d7e04a6650c","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Videollm-mod: Efficient video-language streaming with mixture-of-depths vision computation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:7b75969e1eded840114887e546bf6d98d3b25179ea45dedcd6aa35141b50144c","observation_id":"f4bf46b1-3444-464b-9d89-526ff4284aea","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.03663","last_updated":"2025-03-06T16:25:37Z","snapshot_observed_at":"2026-07-06T20:47:19.219786Z","submitted_at":"2025-03-05T16:52:34Z","title":"LION-FS: Fast & Slow Video-Language Thinker as Online Video Assistant","version":2},"cited_work":{"arxiv_id":"2503.03663","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.03663","snapshot_observed_at":"2026-07-03T20:38:56.133065Z","title":"Lion-fs: Fast & slow video-language thinker as online video assistant,","venue":null,"work_id":"879985b7-119b-48e5-98d7-042daf125c07","year":2025},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"cited_paper":"/paper/2503.03663","citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:11061b71df643aef24650705fcf6da0b7b2d84c629a6205ec996d21ca4b49061","observation_id":"d7174b84-30a2-46e4-9cb7-35dc0e6ed94e","resolution":{"observed_at":"2026-07-02T17:07:12.842378Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06220","last_updated":"2025-09-07T10:23:25Z","snapshot_observed_at":"2026-07-06T20:49:09.368757Z","submitted_at":"2025-03-08T13:44:38Z","title":"StreamMind: Unlocking Full Frame Rate Streaming Video Dialogue through Event-Gated Cognition","version":3},"cited_work":{"arxiv_id":"2503.06220","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06220","snapshot_observed_at":"2026-07-03T20:38:56.187640Z","title":"Stream- mind: Unlocking full frame rate streaming video dialogue through event-gated cognition","venue":null,"work_id":"4d08c82c-d8e5-44d3-9d8a-c371afea6a6e","year":2025},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"cited_paper":"/paper/2503.06220","citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:91b034d6e794ff9d43134a9a00ae87cdcdcabcc50274b95049ea875320aba997","observation_id":"1230205f-9be4-4540-aa55-7e786b5f8d94","resolution":{"observed_at":"2026-07-02T17:07:12.869265Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2411.17991","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T20:48:55.463368Z","title":"Videollm knows when to speak: Enhancing time-sensitive video comprehension with video-text duet interaction format","venue":null,"work_id":"0bc13a71-30dd-466a-974f-09324d2a7814","year":2024},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:019dc8e5c51d2edbaa84ff32468967e9214c5c7e8386e66abf148d237cbbcc21","observation_id":"d66e9752-c647-4405-a5d7-dac30b4e6cbe","resolution":{"observed_at":"2026-07-02T17:07:12.858844Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2505.05467","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-02T17:27:15.523830Z","title":"Streambridge: Turning your offline video large language model into a proactive streaming assistant","venue":null,"work_id":"6f0d7f4a-5862-4b5f-8e70-eeaff1ed8a54","year":2025},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:97534ea9c1a615ae0cfc585bc93b6e165fbc620ff9e5cd1a6912c423f6387f3f","observation_id":"46820132-3a17-4a48-adab-3482f65d276d","resolution":{"observed_at":"2026-07-02T17:07:12.894856Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Streaming long video understanding with large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:202d1e6137091f60e96a6b77d8880571794f6a68da5fcd532c0c0779a1cd84bb","observation_id":"c8ac6f70-b6fa-405c-b10c-9647e0ccc7a4","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Livestar: Live streaming assistant for real-world online video understanding,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:9e4d3328db602b355bef99ef707be5263c0d1613708701faecdd6ce861d98225","observation_id":"f55d66f6-6600-4c4f-bba3-e2673b0c26db","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.19206","last_updated":"2025-05-25T16:11:10Z","snapshot_observed_at":"2026-07-06T21:30:09.658226Z","submitted_at":"2025-05-25T16:11:10Z","title":"SpeakStream: Streaming Text-to-Speech with Interleaved Data","version":1},"cited_work":{"arxiv_id":"2505.19206","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.19206","snapshot_observed_at":"2026-07-02T17:07:12.867079Z","title":"Speakstream: Streaming text-to-speech with interleaved data,","venue":null,"work_id":"a2f61bd2-d49f-4e3f-b19c-11b4f1d488af","year":2025},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"cited_paper":"/paper/2505.19206","citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:a422423841cc86a3034a9abe7786b062186c0f8dddcff72aa9f0d10e4ff2e645","observation_id":"6227ce77-3ec2-40de-9470-32c1250b84ee","resolution":{"observed_at":"2026-07-02T17:07:12.868675Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Soccernet: A scalable dataset for action spotting in soccer videos,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:ed8745e9183f3d6c92b771aa10c2d4cfba3c4cb70f096fd255f7d7673404180e","observation_id":"958d7d40-c1d3-4ae5-bc13-ddf1ee987a11","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Generating live soccer-match commentary from play data,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:8249a59534caca0549bdf834781d11d1b6ee7f4c214567490918b99c93d19188","observation_id":"7c3403eb-813a-4ae9-88b6-6fb758406a17","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Sharegpt4video: Improving video understanding and generation with better captions,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:3b00c656ee812a4435df7f4fac93308ed27233d37b81838721def5042d4990d9","observation_id":"16658cd1-52b2-4661-9f34-52011b235bcd","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16994","last_updated":"2024-04-29T14:52:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-04-25T19:29:55Z","title":"PLLaVA : Parameter-free LLaVA Extension from Images to Videos for Video Dense Captioning","version":2},"cited_work":{"arxiv_id":"2404.16994","doi":"10.48550/arxiv.2404.16994","metadata_source":"pith","pith_arxiv_id":"2404.16994","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PLLaVA : Parameter-free LLaVA Extension from Images to Videos for Video Dense Captioning","venue":"cs.CV","work_id":"8949d4db-20f2-47c1-83a6-fcbe041b62ef","year":2024},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"cited_paper":"/paper/2404.16994","citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:1c00866f9cba11a09f2f22a543ee18e3f00698d80e2b877a940290c82caf9e9d","observation_id":"ddc228c1-894e-424f-87e6-7733a20abc00","resolution":{"observed_at":"2026-07-02T17:07:12.811669Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Video recap: Recursive captioning of hour-long videos,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:65bf605a3e89dd70c61abcf3ae0626629d57166ac1aafd684798fad98b39d46d","observation_id":"8c1932a5-5ea0-49b5-ac6b-e7123ff82066","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-02T11:57:18.735747Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":"2307.09288","doi":"10.24963/ijcai.2025/706","metadata_source":"pith","pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","venue":"cs.CL","work_id":"68a5177f-d644-44c1-bd4f-4e5278c22f5d","year":2023},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:43914568bfdfcca3fcfc17d9b31cb4d6a282ccd68aaadb4012a92cd0c81c4dc3","observation_id":"9fa00143-7797-4935-bf59-024f7f501136","resolution":{"observed_at":"2026-07-02T17:07:12.874439Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11805","last_updated":"2025-05-09T21:04:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-19T02:39:27Z","title":"Gemini: A Family of Highly Capable Multimodal Models","version":5},"cited_work":{"arxiv_id":"2312.11805","doi":"10.1038/nrn2888","metadata_source":"pith","pith_arxiv_id":"2312.11805","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gemini: A Family of Highly Capable Multimodal Models","venue":"cs.CL","work_id":"83f7c85b-3f11-450f-ac0c-64d9745220b2","year":2023},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"cited_paper":"/paper/2312.11805","citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:54efe7635299adc54438695ad8ea4d16844e80bdaec02ddb550f2966cb808de9","observation_id":"5cc267bd-176e-4e9d-b154-c291d58b9451","resolution":{"observed_at":"2026-07-02T17:07:12.850489Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":"2303.08774","doi":"10.1002/tea.20265","metadata_source":"pith","pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GPT-4 Technical Report","venue":"cs.CL","work_id":"b928e041-6991-4c08-8c81-0359e4097c7b","year":2023},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:0569d569f9730c6f09ce07477f1dd7d780bd170048a860f5e4307f21856f8888","observation_id":"5f1f1c72-766d-48f2-a376-cab0b09d49f0","resolution":{"observed_at":"2026-07-02T17:07:12.861724Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Training language models to follow instructions with human feedback,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:3063b3bd81b6b46dc6c830428c7213adfc99ae7f0b6636800f1e53e902c3e779","observation_id":"e07efb0e-149e-450e-8edd-a60cfaac64fc","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Improving language understanding by generative pre-training,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:6d9fe6849f662095e816d315c86fd8affc894195b4dbca8a8b368ced306c97d5","observation_id":"cfcbfc03-9e1e-4175-a08e-de1d81a5fcac","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Ldre: Llm-based divergent reasoning and ensemble for zero-shot composed image retrieval,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:99588f8969b7c3842922d754048b1d17a7f30d42eb3fee38bb0a28b388491860","observation_id":"41c8b56e-334b-40ee-9663-dfe43f28164a","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Vila: On pre-training for visual language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:d7dd08718c72dbcf1f97f319df7274dc6d61074adcc39305ed47f5f33dac792c","observation_id":"4e915c48-e109-4e87-b4c7-b4f0bf5bfbac","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Seman- tic editing increment benefits zero-shot composed image retrieval,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:50f3bd1723fe8f725c9268f2267b88bddd556bf3651001207d41b29523889a1e","observation_id":"e14d055e-86f4-4011-8ddf-4f8ab84db104","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.15747","last_updated":"2023-11-06T12:20:33Z","snapshot_observed_at":"2026-08-06T03:16:21.214943Z","submitted_at":"2023-10-24T11:44:39Z","title":"Large Language Models are Temporal and Causal Reasoners for Video Question Answering","version":2},"cited_work":{"arxiv_id":"2310.15747","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.15747","snapshot_observed_at":"2026-07-03T20:48:55.458276Z","title":"Large language models are temporal and causal reasoners for video question answering","venue":null,"work_id":"6942ee49-a63d-407a-98c0-fc5f721d08d4","year":2023},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"cited_paper":"/paper/2310.15747","citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:1cf16654be832cd1ab388e8452b50976bc000f0347817d7f8b1738f2081dd78b","observation_id":"1c113b12-9f8a-4a48-957e-1f4b72ce12e9","resolution":{"observed_at":"2026-07-02T17:07:12.804217Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Mvbench: A comprehensive multi-modal video understanding benchmark,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:c3db211cda2781639f2bd4883c14792c36887a5a1c15650bc25c978ba97bc98a","observation_id":"0fe125f9-d2ae-4f88-95bf-91beea473d4b","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.09418","last_updated":"2024-06-13T17:59:59Z","snapshot_observed_at":"2026-07-06T18:30:32.982860Z","submitted_at":"2024-06-13T17:59:59Z","title":"VideoGPT+: Integrating Image and Video Encoders for Enhanced Video Understanding","version":1},"cited_work":{"arxiv_id":"2406.09418","doi":"10.48550/arxiv.2406.09418","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.09418","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Videogpt+: Integrating image and video en- coders for enhanced video understanding","venue":"arXiv (Cornell University)","work_id":"b73f5b08-33ad-4759-b16b-0663cf9728b1","year":2024},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"cited_paper":"/paper/2406.09418","citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:2ee7694e61ed476f1f443066dbbf9ac4d247683b7ac6128e0568dee4cab6eb61","observation_id":"23c6ebff-b944-4eb8-94a0-615f5a6f9d25","resolution":{"observed_at":"2026-07-02T17:07:12.831217Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Llava-next: A strong zero-shot video understanding model,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:614ff65ee22ae0f2a5a8428de3bceeb74bc6d148a7f1492f8eb7a700c301ef8e","observation_id":"1f48bc89-8f8f-4cb3-8dd8-cab0f38f4898","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Learning to answer visual questions from web videos,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:8f8909ad795f2b008479147aa82e7812e3ae242bef52fed880d11432e5a7a778","observation_id":"46d4f6c4-d652-4d9f-84d6-d751bdb8cdd6","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Transformer-empowered invariant grounding for video question answering,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:19662a3cd012f67e28ec1872e49710471a569deec060ac25caf81e22778225da","observation_id":"98afd22a-9b1f-4963-9c92-fa20c5be6321","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Intentqa: Intent question answering in videos by cognitive context reasoning,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:408a33643fce07bf0a04cd5f5d1ef862f281d2d9ade72c613e32907b095560e5","observation_id":"49492a6e-29e4-4435-aab6-47ff51297220","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Parse, align and aggregate: Graph-driven compositional reasoning for video question answering,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:b0cba11e2bd4d2a7fd13d342cdeedc328c40550339d0437100c9984d5b5e8d42","observation_id":"5698a124-ebf2-4180-87f2-898289dfe6ad","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Mecd+: Unlocking event-level causal graph discovery for video reasoning,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:cc0980c856963965f77296fe9e6fae81013a6f05c024b19921666b569149b52e","observation_id":"6de7a588-f3bb-4e14-b3b3-5bfb8a2dfec8","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Adversa: Abductive driving accident video understanding,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:efb3e5eef88a71abd06bc378f033e82fc1b3a5d234f455941417702425847c1a","observation_id":"40fc762a-d2ef-4453-ac9b-3b88c89b558f","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Moviechat+: Question-aware sparse memory for long video question answering,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:fd3cf6141ca86e4eb5f467522515b50c36b42fa8c78cee0f9d51bc5ac85985cf","observation_id":"35083534-a016-4ef5-b87c-50f74c1333e8","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Vtg-llm: Integrating timestamp knowledge into video llms for enhanced video temporal grounding,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:c61aae6fccfad23b82083db48daf5f6a8248dd49a52d3b4fcb993b38c9d31f93","observation_id":"85b7a303-a2cd-4634-8cf3-311c71fb059b","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Vtg-gpt: Tuning-free zero- shot video temporal grounding with gpt,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:cf1d964d40cb50df87502ec4496d496f58395012fd1cf52c1c2811c6fbb3b101","observation_id":"a747fe2a-7028-4349-a0c4-b1a75ea58ca7","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.10228","last_updated":"2024-03-15T11:58:18Z","snapshot_observed_at":"2026-07-06T17:45:11.343569Z","submitted_at":"2024-03-15T11:58:18Z","title":"HawkEye: Training Video-Text LLMs for Grounding Text in Videos","version":1},"cited_work":{"arxiv_id":"2403.10228","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.10228","snapshot_observed_at":"2026-07-03T20:38:56.195732Z","title":"arXiv preprint arXiv:2403.10228 , year=","venue":null,"work_id":"fb2ec1aa-a3b0-43e2-9e93-2b73d3097074","year":2024},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"cited_paper":"/paper/2403.10228","citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:e390d7db0c80530bb1a9655e3b231dd6552aa67aa1a50ae5e0b5aff65e9c731d","observation_id":"00a19d4c-3a9f-4e3b-a7f2-bfc08c45edc1","resolution":{"observed_at":"2026-07-02T17:07:12.834984Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"A survey on video temporal grounding with multimodal large language model,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:a470d9fd0db3a507832741c1e4aaf212d7b905387fb6354064ff9f11d82dc943","observation_id":"cec50cc2-e48c-4eac-bb6c-b309438f64c1","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.17421","last_updated":"2023-10-11T05:07:37Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-09-29T17:34:51Z","title":"The Dawn of LMMs: Preliminary Explorations with GPT-4V(ision)","version":2},"cited_work":{"arxiv_id":"2309.17421","doi":"10.48550/arxiv.2309.17421","metadata_source":"pith","pith_arxiv_id":"2309.17421","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"The Dawn of LMMs: Preliminary Explorations with GPT-4V(ision)","venue":"cs.CV","work_id":"344e9dbe-1d9b-4992-a4f1-9bc649978f46","year":2023},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"cited_paper":"/paper/2309.17421","citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:86614ef168fa2f463127ae153138493a6f95cfa6882fb8daba5f23b3339dc81e","observation_id":"7cd82c12-2165-4bd5-9b8d-f56eb15859e9","resolution":{"observed_at":"2026-07-02T17:07:12.873755Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.06261","last_updated":"2025-12-19T14:25:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-07T17:36:04Z","title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","version":6},"cited_work":{"arxiv_id":"2507.06261","doi":"10.48550/arxiv.2503.19","metadata_source":"pith","pith_arxiv_id":"2507.06261","snapshot_observed_at":"2026-07-11T03:17:51.364436Z","title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","venue":"cs.CL","work_id":"008df105-2fdd-45d8-857a-8e35868aecb6","year":2025},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"cited_paper":"/paper/2507.06261","citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:422eaf2c892893e3899237223f93a2d89ba08dac82c0f6e70ade9f5bf1dbb337","observation_id":"6fedf0cf-600e-492c-97bd-cef25aae9926","resolution":{"observed_at":"2026-07-02T17:07:12.884170Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Valor: Vision-audio-language omni-perception pretraining model and dataset,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:146e34701435c23f68a0b3b890598632f136cd62cd36388643dc04214a8125ec","observation_id":"20bb6c12-8282-4aeb-a318-a41cbfcf2743","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Cap4video++: Enhancing video understanding with auxiliary captions,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:06f0676d4ae6265d024589848ec354da73d1e047cefce1d7234a0612b4136644","observation_id":"04f1bc10-b5a7-4c01-b5c6-f2a25f2bcadb","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Hierarchical banzhaf interaction for general video-language representation learning,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:d0d70ba9ff8d0dbcaefbdaeae7c20d71af63ed624fd1b7d5e119db00d84d2dc5","observation_id":"f0ff9915-3ee8-4d2e-9c08-b01d8b655049","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Video dataflywheel: Resolving the impossible data trinity in video-language understanding,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:eb854d194c25ba831c03e3d72e625242dd8e7191540176308bb407da3540f2fd","observation_id":"fe3559d6-e628-47ea-880f-a3e500390de5","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.10122","last_updated":"2024-10-01T12:07:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-11-16T10:59:44Z","title":"Video-LLaVA: Learning United Visual Representation by Alignment Before Projection","version":3},"cited_work":{"arxiv_id":"2311.10122","doi":"10.48550/arxiv.2311.10122","metadata_source":"pith","pith_arxiv_id":"2311.10122","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Video-LLaVA: Learning United Visual Representation by Alignment Before Projection","venue":"cs.CV","work_id":"e2121c51-a55e-476a-af81-7ba6970fe6cf","year":2023},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"cited_paper":"/paper/2311.10122","citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:eb7a03ec1a8cc9aba4bee19384180704cb46bed6d22f1e72debca0840e8ad5b7","observation_id":"307bbdf0-3cc4-4161-b222-78059325826a","resolution":{"observed_at":"2026-07-02T17:07:12.839847Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Streaming dense video captioning,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:1955eb8804235bf4bb2d495ec43f0aa1b6535e94d8b8533e12b56cc598da8ac5","observation_id":"68141cf3-11bd-49eb-98f9-eb56b4648e18","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13468","last_updated":"2025-01-23T08:33:10Z","snapshot_observed_at":"2026-07-06T20:24:50.262766Z","submitted_at":"2025-01-23T08:33:10Z","title":"Streaming Video Understanding and Multi-round Interaction with Memory-enhanced Knowledge","version":1},"cited_work":{"arxiv_id":"2501.13468","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.13468","snapshot_observed_at":"2026-07-04T03:19:29.878882Z","title":"Streaming video under- standing and multi-round interaction with memory-enhanced knowledge","venue":null,"work_id":"05bede93-84f8-4b7b-b6e5-10f6ff02394b","year":2025},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"cited_paper":"/paper/2501.13468","citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:eaa153132e0e3129ce37e7dd3918be66d469be57026faf8ff7208e30c1e106f7","observation_id":"cc8386c5-477b-468e-8966-e97c33317c4c","resolution":{"observed_at":"2026-07-02T17:07:12.826651Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Merlot reserve: Neural script knowledge through vision and language and sound,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:2c11d5d03ce9c511ea7640163856461e40638eae15c3071b1b1a63c48733928a","observation_id":"da088f27-ff86-4e43-bf20-b4a5e92fd45e","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.08401","last_updated":"2023-06-14T09:50:06Z","snapshot_observed_at":"2026-08-02T12:17:31.696131Z","submitted_at":"2023-06-14T09:50:06Z","title":"LiveChat: A Large-Scale Personalized Dialogue Dataset Automatically Constructed from Live Streaming","version":1},"cited_work":{"arxiv_id":"2306.08401","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.08401","snapshot_observed_at":"2026-07-03T20:38:56.192968Z","title":"Livechat: A large- scale personalized dialogue dataset automatically constructed from live streaming,","venue":null,"work_id":"e8a8b9e1-3dfd-4fb2-95c1-770e5202e0b1","year":2023},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"cited_paper":"/paper/2306.08401","citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:6fe0adac01e3d9aaf6f5f96c61aec12ad8c9a73d003814a76877d573f363aff1","observation_id":"beaa738a-70ab-47c2-a3de-2f75721b34f1","resolution":{"observed_at":"2026-07-02T17:07:12.833884Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Ego4d: Around the world in 3,000 hours of egocentric video,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:c629aef66c63a76c4b1561dbd906518e2ebf7fc1e70c24620a2731dc5558209f","observation_id":"c49cf114-8c63-410b-a161-ca270edd949c","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.12769","last_updated":"2025-03-17T03:05:31Z","snapshot_observed_at":"2026-07-06T20:53:41.124209Z","submitted_at":"2025-03-17T03:05:31Z","title":"ViSpeak: Visual Instruction Feedback in Streaming Videos","version":1},"cited_work":{"arxiv_id":"2503.12769","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.12769","snapshot_observed_at":"2026-07-02T19:07:17.481649Z","title":"Vispeak: Visual instruction feedback in streaming videos","venue":null,"work_id":"ff02f96e-6220-439c-a2d7-ebe7d4b79876","year":2025},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"cited_paper":"/paper/2503.12769","citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:a2e37755f845f29daec22e690d4c889a4a953eb349eeb106a0d67e2be9bf6920","observation_id":"52daadf1-6945-4010-b379-dc782656f36c","resolution":{"observed_at":"2026-07-02T17:07:12.853502Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Open-ended hierarchical streaming video understanding with vision language models,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:9fd45365f18bbf4289386218b93fee5e0cc0ede72badb69788221735d36da1c4","observation_id":"3481cff8-ebc7-4199-8d03-3b6f1077b17d","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.01875","last_updated":"2026-04-29T13:44:54Z","snapshot_observed_at":"2026-07-06T22:07:07.294341Z","submitted_at":"2025-08-03T18:15:42Z","title":"StreamAgent: Towards Anticipatory Agents for Streaming Video Understanding","version":4},"cited_work":{"arxiv_id":"2508.01875","doi":null,"metadata_source":"pith","pith_arxiv_id":"2508.01875","snapshot_observed_at":"2026-07-02T17:27:15.714091Z","title":"StreamAgent: Towards Anticipatory Agents for Streaming Video Understanding","venue":"cs.CV","work_id":"129a53a1-987b-419c-a0c3-59c8f376a91d","year":2025},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"cited_paper":"/paper/2508.01875","citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:a22886cf7598901d17f307b895044c3f590b41fee98ec8bf461acaed854d19dd","observation_id":"72d567aa-8864-4a4b-b84a-49523d869cec","resolution":{"observed_at":"2026-07-02T17:07:12.886774Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Learning to respond: A large-scale benchmark and progressive learning framework for trigger-centric online video understanding","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:1c7ca43d53094cf8625211466414a94176a90b82c5249da9321d9eaf1366b5dc","observation_id":"bf29b0ec-3341-4261-93fd-305ba69ab72a","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Egospeak: learning when to speak for egocentric conversational agents in the wild,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:b398fc9fa09227c3aa5a26f09cdc156447d7bed045efa888102dd3493749d0a7","observation_id":"fb1e9b31-2784-4889-9a37-db3130d6c89c","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Streamlined dense video captioning,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:aa3a721fae817de38ae1326545545c9dc95b8d0d4be18ae94943d57025be6a34","observation_id":"201df5d6-37ad-478c-aa4a-a827e73da964","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2603.03447","last_updated":"2026-05-24T13:26:19Z","snapshot_observed_at":"2026-08-06T11:30:58.756664Z","submitted_at":"2026-03-03T19:02:46Z","title":"Proact-VL: A Proactive VideoLLM for Real-Time AI Companions","version":3},"cited_work":{"arxiv_id":"2603.03447","doi":null,"metadata_source":"pith","pith_arxiv_id":"2603.03447","snapshot_observed_at":"2026-07-02T22:37:25.728350Z","title":"Proact-VL: A Proactive VideoLLM for Real-Time AI Companions","venue":"cs.CV","work_id":"bc05d3d0-d49d-4f0e-9fdb-1e43831d3854","year":2026},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"cited_paper":"/paper/2603.03447","citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:ccb0392c4446a0c419b49d697020efc4a6a88e065fcab0801c81b32a50e99cfe","observation_id":"591348b0-58bf-44a0-bab9-92564619ecc5","resolution":{"observed_at":"2026-07-02T17:07:12.845142Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2603.08620","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-02T17:07:12.766858Z","title":"Streamready: Learning what to answer and when in long streaming videos,","venue":null,"work_id":"61f5cb42-34c1-4d30-8a03-54fdd5a30f03","year":2026},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:7995e914d92df472711eabaa62a48231ff857e008d7c567b9c2c754ae32f2568","observation_id":"08702598-86a7-4a93-8be2-466b567e768f","resolution":{"observed_at":"2026-07-02T17:07:12.768417Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2601.10323","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-02T17:07:12.880048Z","title":"Roma: Real-time omni-multimodal assistant with interactive streaming under- standing,","venue":null,"work_id":"377bfb1e-2dfe-4661-9502-1dffb4e767ab","year":2026},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:846824a2e1c1f7e11e972cbcfb97aeb3d13935b2a82f908be6854f4f803ddb53","observation_id":"8aba4993-2c88-4504-998f-06d72778a934","resolution":{"observed_at":"2026-07-02T17:07:12.881699Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2603.27593","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-02T17:07:12.780204Z","title":"Rehg, Minsu Kim, and Yong Man Ro","venue":null,"work_id":"57b923c0-d1d6-4742-a3d8-bf1ddbdaae3d","year":2026},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:2e3a60529cac1caf117e66d138f28fb933c94f0d9c17e11eaa686f70854c6b99","observation_id":"6c7b5c68-8c65-44f2-9aaf-f5cba3b59cf2","resolution":{"observed_at":"2026-07-02T17:07:12.781759Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2603.19054","last_updated":"2026-06-28T06:24:45Z","snapshot_observed_at":"2026-07-13T22:13:00.259166Z","submitted_at":"2026-03-19T15:47:50Z","title":"Em-Garde: A Propose-Match Framework for Proactive Streaming Video Understanding","version":2},"cited_work":{"arxiv_id":"2603.19054","doi":null,"metadata_source":"pith","pith_arxiv_id":"2603.19054","snapshot_observed_at":"2026-07-02T17:07:12.864692Z","title":"Em-Garde: A propose-match framework for proactive streaming video understanding","venue":"cs.CV","work_id":"23ed999f-9308-4f81-ac9e-823cb8ec0aa8","year":2026},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"cited_paper":"/paper/2603.19054","citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:9b4a68c93c153f98220b0410ced938d195bbe2cfdf6f194b5954784b0c110ec1","observation_id":"3820b706-82f9-4baf-9d83-a37a3bc5a889","resolution":{"observed_at":"2026-07-02T17:07:12.865947Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Timechat-online: 80% visual tokens are naturally redundant in streaming videos,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:44a4a02a6fc419684c2ff165975dee9584b3b8d448f34998dd241b6d764a42bf","observation_id":"fbfffffb-f11f-406d-a276-5a27f8c652c1","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Livecc: Learning video llm with streaming speech transcription at scale,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:cf0ea00c26610add1345798c870415553252859a52a8fb51a18592f325ed3a29","observation_id":"19e98f7b-c8bd-4cf1-a610-ef2657e38882","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.14560","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-02T17:27:15.810622Z","title":"Eyes wide open: Ego proactive video-llm for streaming video","venue":null,"work_id":"73840420-3ce8-489d-b9d2-2e7e70fbccbf","year":2025},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:f6b40bf364512a5b82e64fe1982247d5a9ffd233368087c12ddc6a2326a3073f","observation_id":"fde229d1-7ce0-44d1-8369-a33998627afd","resolution":{"observed_at":"2026-07-02T17:07:12.818679Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.05904","last_updated":"2025-06-06T09:23:29Z","snapshot_observed_at":"2026-07-06T21:37:51.892023Z","submitted_at":"2025-06-06T09:23:29Z","title":"Proactive Assistant Dialogue Generation from Streaming Egocentric Videos","version":1},"cited_work":{"arxiv_id":"2506.05904","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.05904","snapshot_observed_at":"2026-07-02T17:07:12.870465Z","title":"person wearing [clothing description]","venue":null,"work_id":"cf176532-0efd-44ab-8327-d512bb5d992c","year":2025},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"cited_paper":"/paper/2506.05904","citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:cad3664f4c7d92e92538ec04319c22b4ca4216ad5f531ededa4002086b45df7e","observation_id":"2f57766a-d5a3-48f5-a834-5e95d888ffed","resolution":{"observed_at":"2026-07-02T17:07:12.871955Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.21904","last_updated":"2025-03-27T18:30:47Z","snapshot_observed_at":"2026-08-05T20:33:30.320758Z","submitted_at":"2025-03-27T18:30:47Z","title":"AssistPDA: An Online Video Surveillance Assistant for Video Anomaly Prediction, Detection, and Analysis","version":1},"cited_work":{"arxiv_id":"2503.21904","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.21904","snapshot_observed_at":"2026-07-02T17:07:12.880276Z","title":"Assistpda: An online video surveillance assistant for video anomaly prediction, detection, and analysis,","venue":null,"work_id":"37c55702-fc95-4180-9a02-67925de01d7f","year":2025},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"cited_paper":"/paper/2503.21904","citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:c0ccc01ae5c01440103aaafb02ff27574f63d1508362a856b7a44d70c5b93705","observation_id":"3389fc50-b9c4-4ea3-97f1-b50c8be4f684","resolution":{"observed_at":"2026-07-02T17:07:12.881882Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2512.21334","last_updated":"2026-04-10T15:00:46Z","snapshot_observed_at":"2026-08-03T08:10:30.527963Z","submitted_at":"2025-12-24T18:59:36Z","title":"Streaming Video Instruction Tuning","version":2},"cited_work":{"arxiv_id":"2512.21334","doi":null,"metadata_source":"pith","pith_arxiv_id":"2512.21334","snapshot_observed_at":"2026-07-03T00:57:30.599699Z","title":"Streaming Video Instruction Tuning","venue":"cs.CV","work_id":"9ee3a429-617c-4c74-87ef-2facd67423f1","year":2025},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"cited_paper":"/paper/2512.21334","citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:b304ae26df0d2149fe5fb3a98b4631c6f78674ff8e02ca51b61c20f87c276205","observation_id":"4c7ddc3f-bec3-44a8-a7ce-90f8d129c2a5","resolution":{"observed_at":"2026-07-02T17:07:12.823683Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2512.06810","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-02T17:07:12.837796Z","title":"Mmduet2: Enhancing proactive interaction of video mllms with multi-turn reinforcement learning","venue":null,"work_id":"f6d90179-5ce6-4788-9ed2-21bafe09531c","year":2025},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:0d32e71319b50a1c435d85353e07837ec1c1f51dba394a34579a22a570807e3d","observation_id":"8fa9d949-2f76-4ff4-83ea-401707a18e0e","resolution":{"observed_at":"2026-07-02T17:07:12.839489Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2603.12938","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-02T17:07:12.890978Z","title":"Thinking in streaming video","venue":null,"work_id":"d76098fe-8c64-4d90-92e4-ce36c34c9ed6","year":2026},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:64248b27e31203fc8528371f9b9db7eaa0ffa9ee9d40d4f55464d89247b9af23","observation_id":"913de0f0-f8c2-4060-b632-aba4ed5a6a36","resolution":{"observed_at":"2026-07-02T17:07:12.892281Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2603.22120","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-02T17:07:12.882757Z","title":"Streamingclaw technical report","venue":null,"work_id":"ee3702bb-9d27-47f0-9723-3e50803224db","year":2026},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:203c98b54e5221dbd75443f19ec2306ea8b90a87ea33aec533b686c8a63c682f","observation_id":"6c894e2e-acde-421f-ab67-e992241ade2e","resolution":{"observed_at":"2026-07-02T17:07:12.884040Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Querystream: Advancing streaming video understanding with query-aware pruning and proactive response,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:308ac6cca1f15ef3df928727970b0e184a6957c00a6a3de5b1968a498c543d16","observation_id":"b10940fc-465e-44b2-8354-eee9e74ef737","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Attention is all you need,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:1c332e42c51b40168231c3c295ff8e6b29f96777f1665ce8982e3576d0222a5e","observation_id":"3f0e521a-1a94-439d-9a8b-4638881663ad","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05530","last_updated":"2024-12-16T17:39:39Z","snapshot_observed_at":"2026-07-06T17:41:42.995949Z","submitted_at":"2024-03-08T18:54:20Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","version":5},"cited_work":{"arxiv_id":"2403.05530","doi":"10.48550/arxiv.2403.05530","metadata_source":"pith","pith_arxiv_id":"2403.05530","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","venue":"cs.CL","work_id":"80e3e977-f1bb-4c83-8d0c-1ab0a0c5c3f1","year":2024},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":96,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"cited_paper":"/paper/2403.05530","citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:d1def0e0cd27573c5376cbdf39c950acf25751cd5da012a1c75e2cc58cc8e433","observation_id":"947f3dfa-9509-4b99-ad6f-88e084244afa","resolution":{"observed_at":"2026-07-02T17:07:12.879057Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-02T03:08:14.426583+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T03:08:14.426583+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:abddd97c04c432950a03566a691c1aceceffc2c459001a155e4476ca9f6d66e7","observation_id":"b62462ba-6f47-43e3-948c-df820a424819","resolution":{"observed_at":"2026-07-02T17:07:12.785043Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.01800","last_updated":"2024-08-03T15:02:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-03T15:02:21Z","title":"MiniCPM-V: A GPT-4V Level MLLM on Your Phone","version":1},"cited_work":{"arxiv_id":"2408.01800","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.01800","snapshot_observed_at":"2026-07-10T11:37:03.161139Z","title":"MiniCPM-V: A GPT-4V Level MLLM on Your Phone","venue":"cs.CV","work_id":"0f06e436-0c76-4e3c-be5e-6168f6bc4336","year":2024},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":98,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"cited_paper":"/paper/2408.01800","citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:8e65095b3f0c0e44d30bb7deb5cfb4d5064948fffc34cce5fba646d25a015b97","observation_id":"1b47b3d4-dc9d-473a-a501-7d4530b185f8","resolution":{"observed_at":"2026-07-02T17:07:12.812326Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.13923","last_updated":"2025-02-19T18:00:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-19T18:00:14Z","title":"Qwen2.5-VL Technical Report","version":1},"cited_work":{"arxiv_id":"2502.13923","doi":"10.48550/arxiv.2502.13923","metadata_source":"pith","pith_arxiv_id":"2502.13923","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen2.5-VL Technical Report","venue":"cs.CV","work_id":"69dffacb-bfe8-442d-be86-48624c60426f","year":2025},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":99,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"cited_paper":"/paper/2502.13923","citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:8314502fe5d96cdc54578697db3e03d58e35fa8b248f84279aa241a81019f615","observation_id":"4fdfc87b-a66e-46e7-8164-46bd3ab84990","resolution":{"observed_at":"2026-07-02T17:07:12.798286Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:19:13.082554+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:19:13.082554+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T22:11:01.690237Z","title":"Dispider: Enabling video llms with active real-time interaction via disentangled perception, decision, and reaction,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":100,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:17ca16dd3ed6e9d28490dbb5f65e7a3e8128aa419d45a74901c5da30b5442501","observation_id":"6942a4ba-250e-4a9a-a0f8-dd6cd29fc844","resolution":{"observed_at":"2026-06-27T22:11:01.690237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-07-06T19:45:57.808548Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":"2411.03628","doi":"10.48550/arxiv.2411.03628","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Streamingbench: Assessing the gap for mllms to achieve streaming video un- derstanding","venue":"arXiv (Cornell University)","work_id":"caec985b-dd2d-4dbb-9199-d147732de99a","year":2024},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":101,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:d4769c57763408d1c788d5f7724210341ff9ffae42d4413e49072e1c7d010302","observation_id":"b275d5f1-12b8-4bb1-a872-6cacc65ccf9d","resolution":{"observed_at":"2026-07-02T17:07:12.871284Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding"},"reference_resolution":{"displayed":100,"state_counts":{"malformed_identifier":0,"metadata_mismatch":3,"parse_uncertain":0,"unresolved":54,"verified_exact":43,"verified_fuzzy":0},"total_outbound_references":120},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 100 of 120 outbound references and 1 inbound Pith citation observation for arXiv:2606.06991."}