{"as_of":"2026-08-22T23:51:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:4b63f66bfb8f1df4975e5e2403374e333b52f17a45538c4636a00c5c03bcef2a","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":48,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":48,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-22T06:32:14.747728+00:00","state":"measured"},{"denominator":48,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":48,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T11:46:40.614557Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":0,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-08-11T16:56:06.254787Z","title":"Streaming- bench: Assessing the gap for mllms to achieve stream- ing video understanding","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.09596","last_updated":"2024-12-12T18:58:30Z","snapshot_observed_at":"2026-08-12T23:49:37.475723Z","submitted_at":"2024-12-12T18:58:30Z","title":"InternLM-XComposer2.5-OmniLive: A Comprehensive Multimodal System for Long-term Streaming Video and Audio Interactions","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-11T16:56:06.254787Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2412.09596"},"observation_digest":"sha256:21b1c781b563a23c1c4410bb73241658444278d64f27ac36e022cd4f81f277f9","observation_id":"9bd4898b-b214-4606-8419-3699604de887","resolution":{"observed_at":"2026-08-11T16:56:06.254787Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":"2411.03628","doi":"10.48550/arxiv.2411.03628","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Streamingbench: Assessing the gap for mllms to achieve streaming video un- derstanding","venue":"arXiv (Cornell University)","work_id":"caec985b-dd2d-4dbb-9199-d147732de99a","year":2024},"citing_paper":{"arxiv_id":"2502.04326","last_updated":"2026-03-01T04:35:41Z","snapshot_observed_at":"2026-08-12T16:10:35.170069Z","submitted_at":"2025-02-06T18:59:40Z","title":"WorldSense: Evaluating Real-world Omnimodal Understanding for Multimodal LLMs","version":3},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-17T05:53:26.066674Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2502.04326"},"observation_digest":"sha256:6e07a61ce6306bf8fac85ed0cf9a31024db7e2bcc60b49c07df89d89d944b53c","observation_id":"6a4356eb-06cf-4207-91f3-aa561ec377ba","resolution":{"observed_at":"2026-05-17T05:53:26.296558Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-08-16T11:46:40.614557Z","title":"Streamingbench: Assessing the gap for mllms to achieve streaming video understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.14693","last_updated":"2025-05-02T22:30:26Z","snapshot_observed_at":"2026-08-18T02:26:00.733954Z","submitted_at":"2025-04-20T17:58:46Z","title":"Video-MMLU: A Massive Multi-Discipline Lecture Understanding Benchmark","version":2},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-08-16T11:46:40.614557Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2504.14693"},"observation_digest":"sha256:f4fc7a3a2301b4ad35f846091307981b36a42b921a43cc8f3602362f87d06b5f","observation_id":"9e7d65b3-1e6a-4799-a3bf-69b740edbb67","resolution":{"observed_at":"2026-08-16T11:46:40.614557Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-08-16T11:16:26.974973Z","title":"Streaming- bench: Assessing the gap for mllms to achieve stream- ing video understanding","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.16030","last_updated":"2025-04-22T16:52:09Z","snapshot_observed_at":"2026-08-19T23:06:03.219257Z","submitted_at":"2025-04-22T16:52:09Z","title":"LiveCC: Learning Video LLM with Streaming Speech Transcription at Scale","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-16T11:16:26.974973Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2504.16030"},"observation_digest":"sha256:5d3b3fbd41f89cbfdab451145f25629bfba4d1bf2c1340f987283833472aeaaa","observation_id":"b345a33e-efb2-405e-b171-865a7a5412c5","resolution":{"observed_at":"2026-08-16T11:16:26.974973Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-08-16T10:47:27.248339Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.17343","last_updated":"2025-04-24T07:59:46Z","snapshot_observed_at":"2026-08-17T17:03:49.803696Z","submitted_at":"2025-04-24T07:59:46Z","title":"TimeChat-Online: 80% Visual Tokens are Naturally Redundant in Streaming Videos","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-16T10:47:27.248339Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2504.17343"},"observation_digest":"sha256:988778d5d433b15b92c22e04026e5149af1035b7afec0b1dbfda300989fb1564","observation_id":"63b43313-834b-48d5-8502-d609831f2d14","resolution":{"observed_at":"2026-08-16T10:47:27.248339Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":"2411.03628","doi":"10.48550/arxiv.2411.03628","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Streamingbench: Assessing the gap for mllms to achieve streaming video un- derstanding","venue":"arXiv (Cornell University)","work_id":"caec985b-dd2d-4dbb-9199-d147732de99a","year":2024},"citing_paper":{"arxiv_id":"2505.07062","last_updated":"2025-05-11T17:28:30Z","snapshot_observed_at":"2026-08-20T20:37:36.775968Z","submitted_at":"2025-05-11T17:28:30Z","title":"Seed1.5-VL Technical Report","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-05-11T05:26:04.960844Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2505.07062"},"observation_digest":"sha256:3dadfca84fe1ab7f01ae4ebeba130ba3c267a230084d73f4894972ec905dd454","observation_id":"a7739048-e734-4807-b2d6-c769a53a1e94","resolution":{"observed_at":"2026-05-11T05:26:05.849502Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-08-15T20:59:44.373465Z","title":"Streamingbench: Assessing the gap for mllms to achieve streaming video understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.11326","last_updated":"2025-05-16T14:48:30Z","snapshot_observed_at":"2026-08-20T14:00:50.097053Z","submitted_at":"2025-05-16T14:48:30Z","title":"Temporally-Grounded Language Generation: A Benchmark for Real-Time Vision-Language Models","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-15T20:59:44.373465Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2505.11326"},"observation_digest":"sha256:a58870df0d6926a4ecd18656429343f02d502bde0f6155d778074c9c6fd1bfe9","observation_id":"d7a362db-5f54-4548-b73e-99ceb56abe63","resolution":{"observed_at":"2026-08-15T20:59:44.373465Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":"2411.03628","doi":"10.48550/arxiv.2411.03628","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Streamingbench: Assessing the gap for mllms to achieve streaming video un- derstanding","venue":"arXiv (Cornell University)","work_id":"caec985b-dd2d-4dbb-9199-d147732de99a","year":2024},"citing_paper":{"arxiv_id":"2505.15269","last_updated":"2026-04-23T12:54:38Z","snapshot_observed_at":"2026-08-12T15:58:40.311449Z","submitted_at":"2025-05-21T08:47:15Z","title":"LiveVLM: Efficient Online Video Understanding via Streaming-Oriented KV Cache and Retrieval","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-22T14:26:59.015559Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2505.15269"},"observation_digest":"sha256:ecf4571e684863f5ea5f03e5d08a19238717a32088fe6f2a3ea3a392e885bdb8","observation_id":"1e5b8677-dccf-4afb-8f70-de7ab238b893","resolution":{"observed_at":"2026-05-22T14:31:40.776721Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-08-15T18:51:52.872171Z","title":"Streamingbench: Assessing the gap for mllms to achieve streaming video understanding, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.18472","last_updated":"2025-06-23T10:11:30Z","snapshot_observed_at":"2026-08-20T06:52:31.435488Z","submitted_at":"2025-06-23T10:11:30Z","title":"AViLA: Asynchronous Vision-Language Agent for Streaming Multimodal Data Interaction","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-15T18:51:52.872171Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2506.18472"},"observation_digest":"sha256:0569edaf6c80785de14ab833f0a47debd68a40e2f4eb1838428e9087f8719f2c","observation_id":"c3ddb6c7-7078-487e-aae5-25a4dc1dd532","resolution":{"observed_at":"2026-08-15T18:51:52.872171Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-08-06T22:41:09.479989Z","title":"Streamingbench: Assessing the gap for mllms to achieve streaming video understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.20960","last_updated":"2025-06-29T15:16:22Z","snapshot_observed_at":"2026-08-16T15:49:09.216949Z","submitted_at":"2025-06-26T02:54:24Z","title":"OmniEval: A Benchmark for Evaluating Omni-modal Models with Visual, Auditory, and Textual Inputs","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T22:41:09.479989Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2506.20960"},"observation_digest":"sha256:702118cf7e50856e26fba46a8ec491baa6ecbe7e45f4583c14102d2785dd54b8","observation_id":"06bc4f0d-0227-4fc5-9358-d442d0b8ea74","resolution":{"observed_at":"2026-08-06T22:41:09.479989Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-08-06T18:03:14.015160Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-06T18:03:14.015160Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2507.09313"},"observation_digest":"sha256:08efc3b212d55d3c3ebcf560b34ac643ff322dc384671bbb58d5f861c8c3fa11","observation_id":"a49e3718-f95e-4026-a5f5-0ac0de340ec5","resolution":{"observed_at":"2026-08-06T18:03:14.015160Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-08-06T05:25:57.420224Z","title":"Jihao Liu, Zhiding Yu, Shiyi Lan, Shihao Wang, Rongyao Fang, Jan Kautz, Hongsheng Li, and Jose M","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.01874","last_updated":"2025-08-03T18:09:40Z","snapshot_observed_at":"2026-08-17T01:16:58.693251Z","submitted_at":"2025-08-03T18:09:40Z","title":"Diffractive electroproduction of light vector particles: leading Fock-state contribution in the presence of significant higher Fock-state effects","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-06T05:25:57.420224Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2508.01874"},"observation_digest":"sha256:1da3f155d52f61a0cd112347bb1da103d4445d6ba0855213cad164c2f2d7a52d","observation_id":"663612ab-4506-4272-a290-c619216eb2f4","resolution":{"observed_at":"2026-08-06T05:25:57.420224Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":"2411.03628","doi":"10.48550/arxiv.2411.03628","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Streamingbench: Assessing the gap for mllms to achieve streaming video un- derstanding","venue":"arXiv (Cornell University)","work_id":"caec985b-dd2d-4dbb-9199-d147732de99a","year":2024},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-08-14T11:36:01.437786Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:57f963f4e89ff3259ad8e265e6147e0cf22907663dfc661326e23994038add8f","observation_id":"9b5af3be-a259-4fad-ac21-759bdd354340","resolution":{"observed_at":"2026-05-17T02:51:28.116389Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-08-02T18:23:40.145298Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.12262","last_updated":"2026-07-17T07:23:56Z","snapshot_observed_at":"2026-08-09T09:42:44.258435Z","submitted_at":"2026-03-12T17:59:51Z","title":"Video Streaming Thinking: VideoLLMs Can Watch and Think Simultaneously","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-02T18:23:40.145298Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2603.12262"},"observation_digest":"sha256:42bf26d238890371b394bb067240545fa6eba693d7a1ff66871b65ddcd271ca2","observation_id":"43fbf30b-2bc3-4141-acc6-7f7e6f52eb0c","resolution":{"observed_at":"2026-08-02T18:23:40.145298Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-07-14T22:09:56.270516Z","title":"arXiv preprint arXiv:2411.03628 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.12703","last_updated":"2026-06-29T17:45:27Z","snapshot_observed_at":"2026-08-16T17:00:00.592893Z","submitted_at":"2026-03-13T06:28:46Z","title":"SVCBench: A Streaming Video Counting Benchmark for Spatial-Temporal State Maintenance","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-07-14T22:09:56.270516Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2603.12703"},"observation_digest":"sha256:b435d6e4bc64713806f26be1a5d1e3881679736c298970d32163c977c7936326","observation_id":"1805d1dd-7ca4-41dd-96db-c66796f2cf02","resolution":{"observed_at":"2026-07-14T22:09:56.270516Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":"2411.03628","doi":"10.48550/arxiv.2411.03628","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Streamingbench: Assessing the gap for mllms to achieve streaming video un- derstanding","venue":"arXiv (Cornell University)","work_id":"caec985b-dd2d-4dbb-9199-d147732de99a","year":2024},"citing_paper":{"arxiv_id":"2603.20633","last_updated":"2026-04-17T08:57:17Z","snapshot_observed_at":"2026-07-06T22:49:57.103822Z","submitted_at":"2026-03-21T04:03:45Z","title":"Seed1.8 Model Card: Towards Generalized Real-World Agency","version":3},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-15T07:44:02.827006Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2603.20633"},"observation_digest":"sha256:c935f73448f8f83f6d843aafd994e327d8562c118009d630c1b86feb8057f1d6","observation_id":"fd5e00ea-18bb-4c81-85b7-c98f527b91a1","resolution":{"observed_at":"2026-05-15T07:45:14.391226Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":"2411.03628","doi":"10.48550/arxiv.2411.03628","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Streamingbench: Assessing the gap for mllms to achieve streaming video un- derstanding","venue":"arXiv (Cornell University)","work_id":"caec985b-dd2d-4dbb-9199-d147732de99a","year":2024},"citing_paper":{"arxiv_id":"2604.07634","last_updated":"2026-05-05T18:37:14Z","snapshot_observed_at":"2026-08-18T17:48:51.144014Z","submitted_at":"2026-04-08T22:31:20Z","title":"VSAS-Bench: Real-Time Evaluation of Visual Streaming Assistant Models","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-10T17:39:30.731688Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2604.07634"},"observation_digest":"sha256:0d3f7ea305de131e0c55c6d9d1f782f20d27c84cdbf821607f8ba2f087b20d96","observation_id":"8141243b-fd9f-4cb1-a5b3-ca88cf3d80cb","resolution":{"observed_at":"2026-05-11T06:25:58.057011Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":"2411.03628","doi":"10.48550/arxiv.2411.03628","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Streamingbench: Assessing the gap for mllms to achieve streaming video un- derstanding","venue":"arXiv (Cornell University)","work_id":"caec985b-dd2d-4dbb-9199-d147732de99a","year":2024},"citing_paper":{"arxiv_id":"2604.11411","last_updated":"2026-04-13T12:55:56Z","snapshot_observed_at":"2026-08-13T17:15:26.490045Z","submitted_at":"2026-04-13T12:55:56Z","title":"Online Reasoning Video Object Segmentation","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-10T15:29:03.843441Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2604.11411"},"observation_digest":"sha256:05e8a51e0d83efdd5d7d881ea40193e7deac43c7a82a8a5cc9515fac05b93652","observation_id":"8fbd045e-016c-4574-bda8-3852d9a6758d","resolution":{"observed_at":"2026-05-11T10:31:00.212771Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":"2411.03628","doi":"10.48550/arxiv.2411.03628","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Streamingbench: Assessing the gap for mllms to achieve streaming video un- derstanding","venue":"arXiv (Cornell University)","work_id":"caec985b-dd2d-4dbb-9199-d147732de99a","year":2024},"citing_paper":{"arxiv_id":"2604.17052","last_updated":"2026-04-18T16:22:05Z","snapshot_observed_at":"2026-08-13T14:22:51.440674Z","submitted_at":"2026-04-18T16:22:05Z","title":"OASIS: On-Demand Hierarchical Event Memory for Streaming Video Reasoning","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-10T06:51:52.861981Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2604.17052"},"observation_digest":"sha256:9f9f477886a18c485d0d06d033407b557f16edea972552b997be2fac9f949df5","observation_id":"5b490176-a817-4edf-8c78-e747aedccbaf","resolution":{"observed_at":"2026-05-10T06:56:47.757004Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-07-12T18:51:26.143516Z","title":"StreamingBench: Assessing the gap for MLLMs to achieve streaming video understanding,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2604.19535","last_updated":"2026-07-08T08:00:17Z","snapshot_observed_at":"2026-08-10T13:10:54.090338Z","submitted_at":"2026-04-21T14:54:24Z","title":"Existence of small semi-vortex solutions for the cubic nonlinear Schr\\\"{o}dinger system with Rashba type Spin-Orbit coupling on $\\mathbb{R}^2$","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-07-12T18:51:26.143516Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2604.19535"},"observation_digest":"sha256:ec2a1cafb924c3d7e5cfc326f4ea9da404002a05e3fa85959a64b5605da0305c","observation_id":"ac1758e3-3689-4543-8d90-2b3341a1d9a9","resolution":{"observed_at":"2026-07-12T18:51:26.143516Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":"2411.03628","doi":"10.48550/arxiv.2411.03628","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Streamingbench: Assessing the gap for mllms to achieve streaming video un- derstanding","venue":"arXiv (Cornell University)","work_id":"caec985b-dd2d-4dbb-9199-d147732de99a","year":2024},"citing_paper":{"arxiv_id":"2604.19536","last_updated":"2026-04-21T14:55:51Z","snapshot_observed_at":"2026-08-10T23:04:08.197347Z","submitted_at":"2026-04-21T14:55:51Z","title":"LiveVLN: Breaking the Stop-and-Go Loop in Vision-Language Navigation","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-10T01:51:23.379734Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2604.19536"},"observation_digest":"sha256:bdb077b10524c6af25881f3f36927ad14ef5f7f081d09113eb768fccf8f3cfdf","observation_id":"e5c4b23d-8c17-4e26-9a5b-4aa2d8ba3853","resolution":{"observed_at":"2026-05-11T13:21:13.476794Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":"2411.03628","doi":"10.48550/arxiv.2411.03628","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Streamingbench: Assessing the gap for mllms to achieve streaming video un- derstanding","venue":"arXiv (Cornell University)","work_id":"caec985b-dd2d-4dbb-9199-d147732de99a","year":2024},"citing_paper":{"arxiv_id":"2604.24317","last_updated":"2026-04-27T11:07:03Z","snapshot_observed_at":"2026-08-21T23:34:15.208367Z","submitted_at":"2026-04-27T11:07:03Z","title":"Don't Pause! Every prediction matters in a streaming video","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-08T04:32:01.379605Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2604.24317"},"observation_digest":"sha256:d18304d49f20e4aa99a61db6f3c8d409bd860b4ee58d06c2623ec033035ee954","observation_id":"38540981-0d22-4fb4-a52e-02067ffebab7","resolution":{"observed_at":"2026-05-11T21:41:17.841108Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":"2411.03628","doi":"10.48550/arxiv.2411.03628","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Streamingbench: Assessing the gap for mllms to achieve streaming video un- derstanding","venue":"arXiv (Cornell University)","work_id":"caec985b-dd2d-4dbb-9199-d147732de99a","year":2024},"citing_paper":{"arxiv_id":"2605.01858","last_updated":"2026-05-03T13:02:44Z","snapshot_observed_at":"2026-08-14T05:50:11.493950Z","submitted_at":"2026-05-03T13:02:44Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-10T14:47:54.917408Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2605.01858"},"observation_digest":"sha256:02b162c5d30b2af7d20dfaabf0b06c256006b33b5e0089a8a31cba64e137f1e7","observation_id":"eb3684ab-300c-44e6-a7d0-7b730bf49737","resolution":{"observed_at":"2026-05-11T11:31:03.575498Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":"2411.03628","doi":"10.48550/arxiv.2411.03628","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Streamingbench: Assessing the gap for mllms to achieve streaming video un- derstanding","venue":"arXiv (Cornell University)","work_id":"caec985b-dd2d-4dbb-9199-d147732de99a","year":2024},"citing_paper":{"arxiv_id":"2605.07575","last_updated":"2026-05-11T11:58:53Z","snapshot_observed_at":"2026-07-06T23:19:56.415523Z","submitted_at":"2026-05-08T10:46:10Z","title":"Response-G1: Explicit Scene Graph Modeling for Proactive Streaming Video Understanding","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-05-11T02:30:55.939351Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2605.07575"},"observation_digest":"sha256:f28fede8d1bdd55df85b7f7b1b9e91b33496885e26b6fd7318fc65ec4b02df4a","observation_id":"4f062cc4-d96a-49cd-93af-920f45b444f0","resolution":{"observed_at":"2026-05-11T03:20:56.399096Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":"2411.03628","doi":"10.48550/arxiv.2411.03628","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Streamingbench: Assessing the gap for mllms to achieve streaming video un- derstanding","venue":"arXiv (Cornell University)","work_id":"caec985b-dd2d-4dbb-9199-d147732de99a","year":2024},"citing_paper":{"arxiv_id":"2605.07575","last_updated":"2026-05-11T11:58:53Z","snapshot_observed_at":"2026-07-06T23:19:56.415523Z","submitted_at":"2026-05-08T10:46:10Z","title":"Response-G1: Explicit Scene Graph Modeling for Proactive Streaming Video Understanding","version":2},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-05-12T03:00:34.728880Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2605.07575"},"observation_digest":"sha256:00538d4cd3dc62f48563d23ce1c52d3c5e75f7cd6bbd39673d55747563070847","observation_id":"8500b5a6-a38e-4324-a067-4887a00535ad","resolution":{"observed_at":"2026-05-12T03:01:17.718872Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":"2411.03628","doi":"10.48550/arxiv.2411.03628","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Streamingbench: Assessing the gap for mllms to achieve streaming video un- derstanding","venue":"arXiv (Cornell University)","work_id":"caec985b-dd2d-4dbb-9199-d147732de99a","year":2024},"citing_paper":{"arxiv_id":"2605.07897","last_updated":"2026-05-08T15:40:40Z","snapshot_observed_at":"2026-08-16T03:45:19.381242Z","submitted_at":"2026-05-08T15:40:40Z","title":"Semantic-Aware Adaptive Visual Memory for Streaming Video Understanding","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-11T02:12:20.651111Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2605.07897"},"observation_digest":"sha256:ff8d2ab985adde25a28f2a1ae04e0569b212aac92208724063bd31e40396035b","observation_id":"66471926-6853-4272-a1f4-7383aab5a4e5","resolution":{"observed_at":"2026-05-11T03:50:56.707856Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":"2411.03628","doi":"10.48550/arxiv.2411.03628","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Streamingbench: Assessing the gap for mllms to achieve streaming video un- derstanding","venue":"arXiv (Cornell University)","work_id":"caec985b-dd2d-4dbb-9199-d147732de99a","year":2024},"citing_paper":{"arxiv_id":"2605.17360","last_updated":"2026-07-02T12:39:55Z","snapshot_observed_at":"2026-08-17T04:09:44.469533Z","submitted_at":"2026-05-17T09:57:01Z","title":"Omni-DuplexEval: Evaluating Real-time Duplex Omni-modal Interaction","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-20T13:36:44.071188Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2605.17360"},"observation_digest":"sha256:0fa561b55e0b3296d68719db7f349c093a02ca0ea5f279757ec5c19d353f462d","observation_id":"1659b41e-c847-4797-be5b-e7baf3a8462d","resolution":{"observed_at":"2026-05-20T13:38:19.141874Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":"2411.03628","doi":"10.48550/arxiv.2411.03628","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Streamingbench: Assessing the gap for mllms to achieve streaming video un- derstanding","venue":"arXiv (Cornell University)","work_id":"caec985b-dd2d-4dbb-9199-d147732de99a","year":2024},"citing_paper":{"arxiv_id":"2605.17360","last_updated":"2026-07-02T12:39:55Z","snapshot_observed_at":"2026-08-17T04:09:44.469533Z","submitted_at":"2026-05-17T09:57:01Z","title":"Omni-DuplexEval: Evaluating Real-time Duplex Omni-modal Interaction","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-07-04T01:11:42.073993Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2605.17360"},"observation_digest":"sha256:62868fa64ee33acd2c6e1f0c73b1a6d08d8cd1a65d3eba92c0dc015755cb0849","observation_id":"034d2306-1167-4490-a1f0-b510d06e5dd4","resolution":{"observed_at":"2026-07-04T01:19:20.357455Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":"2411.03628","doi":"10.48550/arxiv.2411.03628","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Streamingbench: Assessing the gap for mllms to achieve streaming video un- derstanding","venue":"arXiv (Cornell University)","work_id":"caec985b-dd2d-4dbb-9199-d147732de99a","year":2024},"citing_paper":{"arxiv_id":"2605.17921","last_updated":"2026-06-01T06:29:58Z","snapshot_observed_at":"2026-08-15T01:27:26.647958Z","submitted_at":"2026-05-18T06:29:44Z","title":"An Efficient Streaming Video Understanding Framework with Agentic Control","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-05-20T11:30:22.151045Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2605.17921"},"observation_digest":"sha256:bb7b0b0ec0bda8fbb28801d2d60b8a141e39c467968b280d81c0fbe4f45febca","observation_id":"4da3be42-8592-4a21-83ac-9a5735db3fe0","resolution":{"observed_at":"2026-05-20T11:33:14.503209Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":"2411.03628","doi":"10.48550/arxiv.2411.03628","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Streamingbench: Assessing the gap for mllms to achieve streaming video un- derstanding","venue":"arXiv (Cornell University)","work_id":"caec985b-dd2d-4dbb-9199-d147732de99a","year":2024},"citing_paper":{"arxiv_id":"2605.22269","last_updated":"2026-05-21T10:13:03Z","snapshot_observed_at":"2026-08-02T11:53:00.337572Z","submitted_at":"2026-05-21T10:13:03Z","title":"MuKV: Multi-Grained KV Cache Compression for Long Streaming Video Question-Answering","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-22T07:16:33.817183Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2605.22269"},"observation_digest":"sha256:a49e09aee1216e03f8cb7259ebe4b0b58c4aad31187334f9e0f4774905099838","observation_id":"ff030c80-ab6b-4b45-ba45-c3975a3dddef","resolution":{"observed_at":"2026-05-22T07:21:13.153199Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":"2411.03628","doi":"10.48550/arxiv.2411.03628","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Streamingbench: Assessing the gap for mllms to achieve streaming video un- derstanding","venue":"arXiv (Cornell University)","work_id":"caec985b-dd2d-4dbb-9199-d147732de99a","year":2024},"citing_paper":{"arxiv_id":"2605.25621","last_updated":"2026-05-25T09:23:19Z","snapshot_observed_at":"2026-08-15T21:35:32.026398Z","submitted_at":"2026-05-25T09:23:19Z","title":"StreamOV: Streaming Omni-Video Understanding via Evidence-Guided Memory and Response Triggering","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-06-29T22:27:17.092553Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2605.25621"},"observation_digest":"sha256:03a3d86ed5cc0e60fe3015cf2b53f1b20ee6483a46ed0ebc9fdb18aea863688d","observation_id":"ccc60d2c-bc99-4704-94b3-9b9082c406ef","resolution":{"observed_at":"2026-06-29T22:34:02.070102Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":"2411.03628","doi":"10.48550/arxiv.2411.03628","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Streamingbench: Assessing the gap for mllms to achieve streaming video un- derstanding","venue":"arXiv (Cornell University)","work_id":"caec985b-dd2d-4dbb-9199-d147732de99a","year":2024},"citing_paper":{"arxiv_id":"2605.31557","last_updated":"2026-06-01T11:50:06Z","snapshot_observed_at":"2026-08-15T11:32:36.001343Z","submitted_at":"2026-05-29T17:20:10Z","title":"EGOSTREAM: A Diagnostic Benchmark for Streaming Episodic Memory in Egocentric Vision","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-06-28T22:40:13.720341Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2605.31557"},"observation_digest":"sha256:b87f8ae67e94e06b614f54f4838f097ea9e46ade52988f53bad7a4f97545a3ec","observation_id":"154d3637-6094-4768-bc82-77fef37499fa","resolution":{"observed_at":"2026-06-28T22:42:46.378541Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":"2411.03628","doi":"10.48550/arxiv.2411.03628","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Streamingbench: Assessing the gap for mllms to achieve streaming video un- derstanding","venue":"arXiv (Cornell University)","work_id":"caec985b-dd2d-4dbb-9199-d147732de99a","year":2024},"citing_paper":{"arxiv_id":"2606.02482","last_updated":"2026-06-29T07:37:14Z","snapshot_observed_at":"2026-08-14T05:01:18.151983Z","submitted_at":"2026-06-01T16:52:11Z","title":"X-Stream: Exploring MLLMs as Multiplexers for Multi-Stream Understanding","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-06-28T15:06:22.102725Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2606.02482"},"observation_digest":"sha256:0e67ec2fe7f030309fa8233d0df8ac5adb824f10f38b39a70396a297e4801f92","observation_id":"9311291d-1d64-49b3-b8b7-1fe8401b5279","resolution":{"observed_at":"2026-07-01T22:46:18.894558Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":"2411.03628","doi":"10.48550/arxiv.2411.03628","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Streamingbench: Assessing the gap for mllms to achieve streaming video un- derstanding","venue":"arXiv (Cornell University)","work_id":"caec985b-dd2d-4dbb-9199-d147732de99a","year":2024},"citing_paper":{"arxiv_id":"2606.02482","last_updated":"2026-06-29T07:37:14Z","snapshot_observed_at":"2026-08-14T05:01:18.151983Z","submitted_at":"2026-06-01T16:52:11Z","title":"X-Stream: Exploring MLLMs as Multiplexers for Multi-Stream Understanding","version":3},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-06-30T10:42:37.401221Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2606.02482"},"observation_digest":"sha256:2aa127f83bae2485dea47f50aaa01b270b3c9f24830c2534d75a0ea9654f90b6","observation_id":"c5407d40-6258-408b-b3bb-dd295f394749","resolution":{"observed_at":"2026-06-30T10:44:36.402254Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":"2411.03628","doi":"10.48550/arxiv.2411.03628","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Streamingbench: Assessing the gap for mllms to achieve streaming video un- derstanding","venue":"arXiv (Cornell University)","work_id":"caec985b-dd2d-4dbb-9199-d147732de99a","year":2024},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-08-13T16:17:16.712106Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":101,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:171b54fc10d94aa745c4cdc64f4d739796f03863eb6fd5943b877c0e7dbf4ebe","observation_id":"b275d5f1-12b8-4bb1-a872-6cacc65ccf9d","resolution":{"observed_at":"2026-07-02T17:07:12.871284Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":"2411.03628","doi":"10.48550/arxiv.2411.03628","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Streamingbench: Assessing the gap for mllms to achieve streaming video un- derstanding","venue":"arXiv (Cornell University)","work_id":"caec985b-dd2d-4dbb-9199-d147732de99a","year":2024},"citing_paper":{"arxiv_id":"2606.07639","last_updated":"2026-06-01T09:07:15Z","snapshot_observed_at":"2026-08-17T19:24:43.936240Z","submitted_at":"2026-06-01T09:07:15Z","title":"MOSS-Video-Preview: Toward Real-Time Video Understanding via Cross-Attention","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-28T15:22:31.310003Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2606.07639"},"observation_digest":"sha256:174f63449adf25e543b3342036dc523644a380e6c7307fa6c21d57eb876c7bd3","observation_id":"447d702c-84d8-44c7-8173-805dae8e90c5","resolution":{"observed_at":"2026-07-01T22:26:17.981195Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":"2411.03628","doi":"10.48550/arxiv.2411.03628","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Streamingbench: Assessing the gap for mllms to achieve streaming video un- derstanding","venue":"arXiv (Cornell University)","work_id":"caec985b-dd2d-4dbb-9199-d147732de99a","year":2024},"citing_paper":{"arxiv_id":"2606.09547","last_updated":"2026-06-17T20:05:01Z","snapshot_observed_at":"2026-08-07T22:14:54.132288Z","submitted_at":"2026-06-08T14:27:20Z","title":"Streaming Interventions: Can Video Large Language Models Correct Mistakes as They Occur?","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-06-27T16:52:22.811857Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2606.09547"},"observation_digest":"sha256:410135ef625d8deae4854ff2632687ea1d47b7306c0e4bc17d887601f40ce26f","observation_id":"32ecbb8a-0066-4e7c-8ffa-0b790608bbff","resolution":{"observed_at":"2026-07-03T00:57:30.642275Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":"2411.03628","doi":"10.48550/arxiv.2411.03628","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Streamingbench: Assessing the gap for mllms to achieve streaming video un- derstanding","venue":"arXiv (Cornell University)","work_id":"caec985b-dd2d-4dbb-9199-d147732de99a","year":2024},"citing_paper":{"arxiv_id":"2606.22983","last_updated":"2026-06-22T08:04:12Z","snapshot_observed_at":"2026-08-13T19:15:54.888323Z","submitted_at":"2026-06-22T08:04:12Z","title":"LiveServe: Interaction-Aware Serving for Real-Time Omni-Modal LLMs","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-06-26T07:26:07.356352Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2606.22983"},"observation_digest":"sha256:bbc4452dfc69af079aa19c41c8c9772670e9217a74edd9be3f83f52f3509700f","observation_id":"5ce92ae4-e5cd-491d-b047-f8cc06475ea6","resolution":{"observed_at":"2026-06-26T07:29:08.001027Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":"2411.03628","doi":"10.48550/arxiv.2411.03628","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Streamingbench: Assessing the gap for mllms to achieve streaming video un- derstanding","venue":"arXiv (Cornell University)","work_id":"caec985b-dd2d-4dbb-9199-d147732de99a","year":2024},"citing_paper":{"arxiv_id":"2606.24422","last_updated":"2026-06-23T10:59:25Z","snapshot_observed_at":"2026-08-12T17:53:51.025868Z","submitted_at":"2026-06-23T10:59:25Z","title":"EgoSAT: A Comprehensive Benchmark of Egocentric Streaming Interaction Understanding","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-06-26T00:26:33.306128Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2606.24422"},"observation_digest":"sha256:29992e1c7bbb2742e34ecfa54372b713b072ab5853f1c178b53d319dd47d7590","observation_id":"ce35343c-1392-4842-8e43-433ed68de263","resolution":{"observed_at":"2026-07-04T16:39:57.329945Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":"2411.03628","doi":"10.48550/arxiv.2411.03628","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Streamingbench: Assessing the gap for mllms to achieve streaming video un- derstanding","venue":"arXiv (Cornell University)","work_id":"caec985b-dd2d-4dbb-9199-d147732de99a","year":2024},"citing_paper":{"arxiv_id":"2606.26762","last_updated":"2026-06-25T08:47:15Z","snapshot_observed_at":"2026-07-07T00:01:05.946912Z","submitted_at":"2026-06-25T08:47:15Z","title":"ProtoKV: Streaming Video Understanding under Delayed Query with Summary-State Memory","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-06-26T05:18:38.341283Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2606.26762"},"observation_digest":"sha256:d1647de856be6d5d396b0aa4ed4980c39421fab086456c7fe1f8a8bebfb1fe79","observation_id":"3f626472-c382-4986-a0cf-dcec7018dadd","resolution":{"observed_at":"2026-07-04T13:19:50.783334Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":"2411.03628","doi":"10.48550/arxiv.2411.03628","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Streamingbench: Assessing the gap for mllms to achieve streaming video un- derstanding","venue":"arXiv (Cornell University)","work_id":"caec985b-dd2d-4dbb-9199-d147732de99a","year":2024},"citing_paper":{"arxiv_id":"2607.01751","last_updated":"2026-07-02T06:07:44Z","snapshot_observed_at":"2026-08-13T07:31:33.095252Z","submitted_at":"2026-07-02T06:07:44Z","title":"MedStreamBench: A Time-Aware Benchmark for Streaming and Proactive Medical Video Understanding","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-07-03T16:37:09.666491Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2607.01751"},"observation_digest":"sha256:a431b45de5a5619167e159bc47e21e5718c5c095541d87512d0b5fe766fa213a","observation_id":"4cd37fb4-85c1-4265-bc0c-a7e2aabd1774","resolution":{"observed_at":"2026-07-03T16:38:39.538176Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-07-12T05:36:40.278384Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.02991","last_updated":"2026-07-03T06:00:04Z","snapshot_observed_at":"2026-08-06T10:48:23.317748Z","submitted_at":"2026-07-03T06:00:04Z","title":"GuideMe: Multi-Domain Task Guidance and Intervention in Streaming Video","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-07-12T05:36:40.278384Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2607.02991"},"observation_digest":"sha256:c9f6cedccd64767c6c0319e5fa338464f8108d4c49cdbed7b5538d0a0be8c9dd","observation_id":"8d020d31-bed5-447b-a79b-05adf1fa0c91","resolution":{"observed_at":"2026-07-12T05:36:40.278384Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-07-11T17:18:41.284513Z","title":"arXiv preprint arXiv:2411.03628 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.04559","last_updated":"2026-07-06T00:19:42Z","snapshot_observed_at":"2026-08-09T14:49:10.281004Z","submitted_at":"2026-07-06T00:19:42Z","title":"QSVideo: Query-Conditioned Semantic Temporal Retrieval for Video Understanding","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-07-11T17:18:41.284513Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2607.04559"},"observation_digest":"sha256:cf9cb41f2ce93ef3476d928e1607c90bb6eb7c4983344e671986cc2ac847a499","observation_id":"ab925452-f670-4ef3-848f-743a13b195e2","resolution":{"observed_at":"2026-07-11T17:18:41.284513Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-07-14T05:01:06.200663Z","title":"arXiv preprint arXiv:2411.03628 (2024) 15","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.11523","last_updated":"2026-07-13T13:09:45Z","snapshot_observed_at":"2026-08-15T01:01:51.688250Z","submitted_at":"2026-07-13T13:09:45Z","title":"Vinci2: Providing Proactive Assistance in Continuous Egocentric Videos","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-07-14T05:01:06.200663Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2607.11523"},"observation_digest":"sha256:8684bb21f8b007958f87727bbf2b53d42a9c7163d6da1645c97c084d61971c60","observation_id":"95cf5144-b056-45b2-9ea3-86c4e8205783","resolution":{"observed_at":"2026-07-14T05:01:06.200663Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-08-02T01:31:32.083457Z","title":"Streamingbench: Assessing the gap for mllms to achieve streaming video un- derstanding.arXiv preprint arXiv:2411.03628, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.14660","last_updated":"2026-07-16T07:25:17Z","snapshot_observed_at":"2026-08-18T18:19:17.919071Z","submitted_at":"2026-07-16T07:25:17Z","title":"VIABench: A Comprehensive Video Benchmark Collected from Blind Individuals for Visual Impairment Assistance","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-02T01:31:32.083457Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2607.14660"},"observation_digest":"sha256:d89ac4731fdf91db2d62b7bc50c2b6b3edb2f5db9f63fd0973ce82538ca5cc6f","observation_id":"9c3daa94-44a1-42f8-9aef-9bfe00ebec93","resolution":{"observed_at":"2026-08-02T01:31:32.083457Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-07-31T06:20:13.909328Z","title":"StreamingBench: Assessing the gap for mllms to achieve streaming video understanding.arXiv preprint arXiv:2411.03628, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24904","last_updated":"2026-07-27T17:59:53Z","snapshot_observed_at":"2026-08-15T01:08:05.619745Z","submitted_at":"2026-07-27T17:59:53Z","title":"Mage-VL: An Efficient Codec-Native Streaming Multimodal Foundation Model","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-07-31T06:20:13.909328Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2607.24904"},"observation_digest":"sha256:0f82c7d545b287e8e4da51d4bab25285b5ea0c04901cf019091b7f253e49efbb","observation_id":"6c5d6ee0-5824-4d04-8018-5c7fb7c2ab28","resolution":{"observed_at":"2026-07-31T06:20:13.909328Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-08-04T03:21:44.907376Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.28312","last_updated":"2026-08-01T16:00:11Z","snapshot_observed_at":"2026-08-21T19:05:25.712209Z","submitted_at":"2026-07-30T14:47:00Z","title":"ObjectStream: Latent Objects as Memory Anchors for Streaming Video Understanding","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-04T03:21:44.907376Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2607.28312"},"observation_digest":"sha256:4e5bfe3d71a5eebb36461bf28bf46c3ce7238c55a1772b7e3b7f32f162c73a58","observation_id":"dc85cc35-f2f5-44d9-9313-a0455b94904d","resolution":{"observed_at":"2026-08-04T03:21:44.907376Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-08-04T17:24:14.991460Z","title":"2411.03628 , archivePrefix =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.01980","last_updated":"2026-08-03T09:42:42Z","snapshot_observed_at":"2026-08-17T18:20:09.005534Z","submitted_at":"2026-08-03T09:42:42Z","title":"AdaThinkV: Adaptive Thinking for Token-Efficient Video Reasoning","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-04T17:24:14.991460Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2608.01980"},"observation_digest":"sha256:88bf3ebce15055037c1ee9fbf2adf0d6ab1c8c828e492c67768f5969e081eb47","observation_id":"5dcb945f-5939-47d5-8b86-9410a0d91304","resolution":{"observed_at":"2026-08-04T17:24:14.991460Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2411.03628/citation-record","integrity":"/paper/2411.03628/integrity","json":"/paper/2411.03628/citation-record.json","paper":"/paper/2411.03628"},"outbound":[],"paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-20T14:00:31.960419Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"thesis":"As of 22 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 48 inbound Pith citation observations for arXiv:2411.03628."}