{"as_of":"2026-08-05T22:29:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:507d6a319794ae8e65fe963d7178dd73f02eaf81023ee2f1b167507735f0c7b3","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":43,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":43,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-05T06:32:48.257954+00:00","state":"measured"},{"denominator":43,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":43,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T08:12:47.458517Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T20:00:08.182505Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2408.10188","last_updated":"2024-12-13T02:32:06Z","snapshot_observed_at":"2026-08-05T14:57:53.592979Z","submitted_at":"2024-08-19T17:48:08Z","title":"LongVILA: Scaling Long-Context Visual Language Models for Long Videos","version":6},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-17T03:51:25.396887Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2408.10188"},"observation_digest":"sha256:a40b86f014cf1ff6abae71bdd0ad29e35360f879e0ebf3f4ffa24cff0b7a9d08","observation_id":"16ce91fa-49dc-4fb9-85aa-a081ec43c810","resolution":{"observed_at":"2026-05-17T03:51:25.463933Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2411.02327","last_updated":"2026-05-01T17:44:24Z","snapshot_observed_at":"2026-07-06T19:45:00.034364Z","submitted_at":"2024-11-04T17:50:36Z","title":"PPLLaVA: Varied Video Sequence Understanding With Prompt Guidance","version":4},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-23T17:31:59.030963Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2411.02327"},"observation_digest":"sha256:11ba3d1305994e554e77daaf4ccf233de59a3f917dd81d3fa7347c4b47b3b4d2","observation_id":"fed152d1-acea-4a91-8536-678bbe328e50","resolution":{"observed_at":"2026-05-23T17:33:15.672665Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2501.13106","last_updated":"2025-06-03T03:33:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T18:59:46Z","title":"VideoLLaMA 3: Frontier Multimodal Foundation Models for Image and Video Understanding","version":4},"reference_index":168,"source":"pdf_text","source_observed_at":"2026-05-11T01:19:59.603343Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2501.13106"},"observation_digest":"sha256:792b4e9cddc829df252f881e762c204bcfa5dab82516ab01a04b0b96f41cea43","observation_id":"80098db2-8922-4721-9090-bfaa8ada79d8","resolution":{"observed_at":"2026-05-11T01:20:00.236233Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2505.15269","last_updated":"2026-04-23T12:54:38Z","snapshot_observed_at":"2026-08-03T01:07:06.468931Z","submitted_at":"2025-05-21T08:47:15Z","title":"LiveVLM: Efficient Online Video Understanding via Streaming-Oriented KV Cache and Retrieval","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-22T14:26:59.015559Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2505.15269"},"observation_digest":"sha256:279436fb015b558edf6c5fa76610facd4bbaa60c637ea26f32be67392fa5be9e","observation_id":"9c3d523b-1f4c-4892-95fa-71444da65956","resolution":{"observed_at":"2026-05-22T14:31:40.813023Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2505.18719","last_updated":"2025-05-24T14:42:51Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-24T14:42:51Z","title":"VLA-RL: Towards Masterful and General Robotic Manipulation with Scalable Reinforcement Learning","version":1},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-05-16T12:55:40.245908Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2505.18719"},"observation_digest":"sha256:0d6db536f7a72b89479dc9846c1317b94eab8add63a038f6fe6a9127fc05c06b","observation_id":"f614c2f9-c1ee-4bd9-92c9-241b1d2d38bb","resolution":{"observed_at":"2026-05-16T12:55:40.390683Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-08-03T20:15:36.429995Z","title":"Flash-vstream: Memory- based real-time understanding for long video streams.arXiv preprint arXiv:2406.08085, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2511.20644","last_updated":"2026-07-09T17:59:10Z","snapshot_observed_at":"2026-08-03T20:15:29.231318Z","submitted_at":"2025-11-25T18:59:02Z","title":"Vision-Language Memory for Spatial Reasoning","version":2},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-03T20:15:36.429995Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2511.20644"},"observation_digest":"sha256:5118e9e67955ca0ec34fecef022c6c71b25c803cab96b59a852e027c32fcff38","observation_id":"946eac70-1f6c-4433-aad0-c82da8c1daea","resolution":{"observed_at":"2026-08-03T20:15:36.429995Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2511.21998","last_updated":"2026-04-12T05:29:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-11-27T00:54:35Z","title":"Can Multi-Modal LLMs Provide Live Step-by-Step Task Guidance?","version":2},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-05-17T05:36:09.208754Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2511.21998"},"observation_digest":"sha256:4bb57831039cec69c18b5157a33a779826fd1bc44f17f7262d976c98b43092ba","observation_id":"14008280-962a-4104-a8cc-86e1bb31ca92","resolution":{"observed_at":"2026-05-17T05:39:06.167340Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2512.21334","last_updated":"2026-04-10T15:00:46Z","snapshot_observed_at":"2026-08-03T08:10:30.527963Z","submitted_at":"2025-12-24T18:59:36Z","title":"Streaming Video Instruction Tuning","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-16T19:44:11.032898Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2512.21334"},"observation_digest":"sha256:e1eaef07e122b064c14afca174d94042e8ab3b6b36622adacc7e59c200522d38","observation_id":"394e8922-e627-4b86-b997-907e05b8b432","resolution":{"observed_at":"2026-05-16T19:48:21.792961Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2601.14724","last_updated":"2026-05-07T12:10:26Z","snapshot_observed_at":"2026-08-05T03:02:47.238051Z","submitted_at":"2026-01-21T07:26:15Z","title":"HERMES: KV Cache as Hierarchical Memory for Efficient Streaming Video Understanding","version":4},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-16T12:55:04.564442Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2601.14724"},"observation_digest":"sha256:0a1eed1a94bd3a77a1b7fbcfbb6c4c50e65cd374819a4deac16b01990b261ab3","observation_id":"22a96c46-ffa6-459c-a1f9-b95b727ad2cb","resolution":{"observed_at":"2026-05-16T12:57:53.864829Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2602.20913","last_updated":"2026-04-15T16:09:22Z","snapshot_observed_at":"2026-08-02T12:38:41.181077Z","submitted_at":"2026-02-24T13:49:47Z","title":"LongVideo-R1: Smart Navigation for Low-cost Long Video Understanding","version":2},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-05-15T20:01:31.129959Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2602.20913"},"observation_digest":"sha256:9939a3b27228de53349316231ac49f17d1cea5639f8ca61a3159966cf9aff9f6","observation_id":"0bf7a826-66ac-4eb0-a31b-6e0dc62ccb89","resolution":{"observed_at":"2026-05-15T20:01:33.505455Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2603.01455","last_updated":"2026-04-21T05:06:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-03-02T05:12:45Z","title":"From Verbatim to Gist: Distilling Pyramidal Multimodal Memory via Semantic Information Bottleneck for Long-Horizon Video Agents","version":3},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-15T18:09:59.236030Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2603.01455"},"observation_digest":"sha256:eab8216b7202bd42d55c172a6083e92d39ae3e26b3fdc54d617965f0d652457a","observation_id":"5bc8b24f-ac61-4f4b-a48e-e571ca4389d6","resolution":{"observed_at":"2026-05-15T18:10:13.107496Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-14T22:25:30.596294Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2603.12219","last_updated":"2026-06-22T03:37:19Z","snapshot_observed_at":"2026-07-14T22:25:30.331088Z","submitted_at":"2026-03-12T17:44:27Z","title":"An Updated SynthPop Model for Microlensing Simulations I: Model Description & Evaluation","version":3},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-07-14T22:25:30.596294Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2603.12219"},"observation_digest":"sha256:51974fb0d53367a42c65d21d9b13afdb01151fb4245bdbc31dcf8754eafd273d","observation_id":"104221c3-5d21-4f4e-a341-9336cef722da","resolution":{"observed_at":"2026-07-14T22:25:30.596294Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2603.27259","last_updated":"2026-06-18T21:01:40Z","snapshot_observed_at":"2026-08-02T11:52:58.572026Z","submitted_at":"2026-03-28T12:44:19Z","title":"Seeing the Scene Matters: Revealing Forgetting in Video Understanding Models with a Scene-Aware Long-Video Benchmark","version":2},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-05-14T22:05:07.326202Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2603.27259"},"observation_digest":"sha256:3bb0845f020659a1132fa3bceba0c787c1db6ae20e6cef4e4879d0c9ecb79b97","observation_id":"df9ac8cf-62b0-49a1-b05c-90746584bb94","resolution":{"observed_at":"2026-05-14T22:08:04.448097Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2604.07634","last_updated":"2026-05-05T18:37:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-08T22:31:20Z","title":"VSAS-Bench: Real-Time Evaluation of Visual Streaming Assistant Models","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-10T17:39:30.731688Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2604.07634"},"observation_digest":"sha256:cd59d9bcd57eba7a85c0966cc8d2354f9b62da04a9d40f19e2d062385036e6be","observation_id":"2583ddc4-2dc7-4abc-af36-05ebbeaae521","resolution":{"observed_at":"2026-05-11T06:25:58.031174Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2604.09000","last_updated":"2026-04-23T08:05:34Z","snapshot_observed_at":"2026-07-06T22:57:59.555972Z","submitted_at":"2026-04-10T06:11:34Z","title":"StreamMeCo: Long-Term Agent Memory Compression for Efficient Streaming Video Understanding","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-10T18:19:41.543165Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2604.09000"},"observation_digest":"sha256:95669abc117e4567f5aada02f189de8d80dd0fdfd853d28fcd912b25bb403525","observation_id":"8cb884b5-c0e4-4c76-b400-7aa8cb670cf2","resolution":{"observed_at":"2026-05-11T00:45:50.436845Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2604.17052","last_updated":"2026-04-18T16:22:05Z","snapshot_observed_at":"2026-08-02T06:02:58.067734Z","submitted_at":"2026-04-18T16:22:05Z","title":"OASIS: On-Demand Hierarchical Event Memory for Streaming Video Reasoning","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-10T06:51:52.861981Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2604.17052"},"observation_digest":"sha256:b048117257c595f1e4a354b2c8e0b5c4beb1f5e6e7de1d783e9d7db6ade17c5a","observation_id":"266777e1-7e61-40fb-aea8-674cb00bbeca","resolution":{"observed_at":"2026-05-10T06:56:47.788325Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2604.24317","last_updated":"2026-04-27T11:07:03Z","snapshot_observed_at":"2026-07-06T23:10:24.992210Z","submitted_at":"2026-04-27T11:07:03Z","title":"Don't Pause! Every prediction matters in a streaming video","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-05-08T04:32:01.379605Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2604.24317"},"observation_digest":"sha256:7db4b0f3cab3736cc80abe56e6eaf5918b71728df9355266f377272440306bec","observation_id":"13efc00b-3054-4d1d-8fd7-92eba1529fe7","resolution":{"observed_at":"2026-05-11T21:41:18.019791Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2605.01858","last_updated":"2026-05-03T13:02:44Z","snapshot_observed_at":"2026-07-06T23:15:01.968194Z","submitted_at":"2026-05-03T13:02:44Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-10T14:47:54.917408Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2605.01858"},"observation_digest":"sha256:5dd2cbc75bb0417a943f37973b27a4aaeefac4b93b46b979b01ac6e18bbde725","observation_id":"09c9c0bb-1fff-461f-aa8e-83141dc06e75","resolution":{"observed_at":"2026-05-11T11:31:03.562532Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2605.04515","last_updated":"2026-05-06T05:48:56Z","snapshot_observed_at":"2026-07-06T23:17:18.486586Z","submitted_at":"2026-05-06T05:48:56Z","title":"From Priors to Perception: Grounding Video-LLMs in Physical Reality","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-08T17:41:23.233366Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2605.04515"},"observation_digest":"sha256:7d1b806e12b32e470f34c09707d786bcb3e51a55c0fcfb94eb11d0fd2eb89590","observation_id":"5cdcc589-c2b1-4b8e-8faa-0b46fd2d4312","resolution":{"observed_at":"2026-05-11T17:21:08.334591Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2605.15054","last_updated":"2026-05-14T16:48:03Z","snapshot_observed_at":"2026-07-06T23:26:23.179482Z","submitted_at":"2026-05-14T16:48:03Z","title":"LATERN: Test-Time Context-Aware Explainable Video Anomaly Detection","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-06-30T21:19:12.655706Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2605.15054"},"observation_digest":"sha256:c5ef448e37e2ab155191e81bbf84bb387a8f9bf7d0712a4c7be1a1fdfdb98042","observation_id":"212f5093-1618-4c58-a2ad-0068af94c72a","resolution":{"observed_at":"2026-07-01T14:25:47.016973Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2605.17065","last_updated":"2026-05-16T16:15:59Z","snapshot_observed_at":"2026-08-02T03:22:29.983512Z","submitted_at":"2026-05-16T16:15:59Z","title":"PyraVid: Hierarchical Multimodal Memory for Long-Horizon Video Reasoning","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-05-20T15:12:00.408851Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2605.17065"},"observation_digest":"sha256:0671084634540ec9b0a9e7d3bc38b8b89c5121a57022b3a4b1ed6ba9bd720a6a","observation_id":"eb91be19-68ab-41e9-ba79-cff4ed9f0561","resolution":{"observed_at":"2026-05-20T15:13:24.798248Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2605.17283","last_updated":"2026-05-17T06:39:05Z","snapshot_observed_at":"2026-08-02T16:55:37.869194Z","submitted_at":"2026-05-17T06:39:05Z","title":"OProver: A Unified Framework for Agentic Formal Theorem Proving","version":1},"reference_index":80,"source":"arxiv_source","source_observed_at":"2026-05-20T14:43:46.517807Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2605.17283"},"observation_digest":"sha256:23f09a8357243fcf1f6a466766c7d936f3ad875522ffb0478bbe7b8a27af7e0a","observation_id":"0c263db3-57e5-4683-92be-a303128f8bbc","resolution":{"observed_at":"2026-05-20T14:48:23.506925Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2605.17360","last_updated":"2026-07-02T12:39:55Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T09:57:01Z","title":"Omni-DuplexEval: Evaluating Real-time Duplex Omni-modal Interaction","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-20T13:36:44.071188Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2605.17360"},"observation_digest":"sha256:4aea4511434a75734d16e250339b095f2c05c86320626b3248bbae1558b8327e","observation_id":"96f99809-c799-4cf6-8451-346d8e4bcfc1","resolution":{"observed_at":"2026-05-20T13:38:19.169529Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2605.17360","last_updated":"2026-07-02T12:39:55Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T09:57:01Z","title":"Omni-DuplexEval: Evaluating Real-time Duplex Omni-modal Interaction","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-07-04T01:11:42.073993Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2605.17360"},"observation_digest":"sha256:5bec1f4922b09b1e5cdb7418df8fd39e190cf8967d8bf917940db65fa4237d51","observation_id":"aaf4586a-8910-4338-80b9-98681305850c","resolution":{"observed_at":"2026-07-04T01:19:20.326643Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2605.27074","last_updated":"2026-05-26T14:23:25Z","snapshot_observed_at":"2026-08-02T20:44:04.329111Z","submitted_at":"2026-05-26T14:23:25Z","title":"IPIBench: Evaluating Interactive Proactive Intelligence of MLLMs under Continuous Streams","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-06-29T18:24:57.881644Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2605.27074"},"observation_digest":"sha256:f325bb3bd5122a061d021df86d2f00f6805609b9d3906ea21b6b3a8bd16622c2","observation_id":"6743be0a-b7b7-40fd-9707-80343302ea06","resolution":{"observed_at":"2026-06-29T18:33:51.028063Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2605.31598","last_updated":"2026-05-29T17:59:02Z","snapshot_observed_at":"2026-07-06T23:40:46.825645Z","submitted_at":"2026-05-29T17:59:02Z","title":"Linear Scaling Video VLMs for Long Video Understanding","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-06-28T23:00:11.246232Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2605.31598"},"observation_digest":"sha256:1fd21afb35d123c71637439736778c80555d4519030b3aa1b26201d28726cab6","observation_id":"2b18fe52-a846-4236-b189-fc792d0a4fb2","resolution":{"observed_at":"2026-06-28T23:02:46.258016Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2606.03100","last_updated":"2026-06-04T05:13:15Z","snapshot_observed_at":"2026-08-01T16:37:35.015535Z","submitted_at":"2026-06-02T03:38:51Z","title":"Zero-Shot 3D Question Answering via Hierarchical View-to-Token Transportation","version":2},"reference_index":81,"source":"arxiv_source","source_observed_at":"2026-06-28T10:54:02.188634Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2606.03100"},"observation_digest":"sha256:b5df5c38e387acab578dc9a1667fecd20bbeff70811e2f7a1c5f13a8b67c3d2f","observation_id":"7356d735-315d-4c30-af27-76028d7db25c","resolution":{"observed_at":"2026-07-02T02:26:27.065722Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2606.03890","last_updated":"2026-06-02T16:51:32Z","snapshot_observed_at":"2026-07-06T23:44:05.167767Z","submitted_at":"2026-06-02T16:51:32Z","title":"OVO-S-Bench: A Hierarchical Benchmark for Streaming Spatial Intelligence in Multimodal LLMs","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-06-28T11:02:07.122615Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2606.03890"},"observation_digest":"sha256:791d5a30b368b3ee5b14f54bf0944072cb954f7c040c734036d825f2b201ff92","observation_id":"cb654ab0-e384-4fd2-9d22-52721b12820f","resolution":{"observed_at":"2026-07-02T02:16:27.100649Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:a842285eb2b8f86dd44fea0167d541db43495f8841d933cb125b38e1b39d871c","observation_id":"f17c9624-dcc6-4b32-9028-2268977d50fc","resolution":{"observed_at":"2026-07-02T17:07:12.826114Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:1b1bda40f89ddc8e7d4e858958ef0749f421ab844a5db4a1633545251d5609ad","observation_id":"ace06fdb-019e-44c5-aecf-8cefd999604a","resolution":{"observed_at":"2026-07-02T17:27:15.497535Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2606.09547","last_updated":"2026-06-17T20:05:01Z","snapshot_observed_at":"2026-08-03T18:46:56.762672Z","submitted_at":"2026-06-08T14:27:20Z","title":"Streaming Interventions: Can Video Large Language Models Correct Mistakes as They Occur?","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-06-27T16:52:22.811857Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2606.09547"},"observation_digest":"sha256:5b77ec35be5b6e7b12a0b133fa0047f5ef6eb54286c44d32bb1326fda3568ef9","observation_id":"8b545684-c27d-47da-8667-b0cb0d11a8eb","resolution":{"observed_at":"2026-07-03T00:57:30.647830Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:f3d4df3b94fe1a1e736b19ca65835bf9d64544c65d24d6c0f9865cbd02610fd6","observation_id":"8b069b59-ddbf-4ffd-8919-b3520863a203","resolution":{"observed_at":"2026-07-03T20:38:56.143340Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2606.19849","last_updated":"2026-06-18T06:57:31Z","snapshot_observed_at":"2026-08-02T14:00:57.631473Z","submitted_at":"2026-06-18T06:57:31Z","title":"ViCoStream: Streaming VideoLLMs Can Run Beyond 100 FPS with Stage-Wise Coordinated Inference","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-06-26T18:17:53.013043Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2606.19849"},"observation_digest":"sha256:bef703db66d7941c0eb332d377ae8ad9a9dd974bc3323641d1ed128ee1662b6c","observation_id":"587d4ed6-2de8-4cf1-aff5-54b31c752bf1","resolution":{"observed_at":"2026-07-04T03:19:29.905981Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2606.20726","last_updated":"2026-06-17T03:30:01Z","snapshot_observed_at":"2026-08-04T00:48:28.857099Z","submitted_at":"2026-06-17T03:30:01Z","title":"How Well Can Your Video Model Remember? Measuring Memory-Budget Trade-offs in Long Video Understanding","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-06-26T21:51:04.050833Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2606.20726"},"observation_digest":"sha256:3ca25f78c18749a02ed012ab8bcf9e076a3f90f989bf225257b4941930263b1d","observation_id":"25d70545-f7d4-4401-ad83-bd7f31670d73","resolution":{"observed_at":"2026-07-03T23:39:04.710461Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2606.24477","last_updated":"2026-06-23T12:13:19Z","snapshot_observed_at":"2026-07-06T23:59:03.724013Z","submitted_at":"2026-06-23T12:13:19Z","title":"video-SALMONN-R$^3$: Learning to ReWatch, ReAsk, and ReAnswer for Efficient Video Understanding","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-06-26T00:19:26.153682Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2606.24477"},"observation_digest":"sha256:f118d7a76e5cf1a5b39cdcc3352481a2ba73ec978b6ad4dc2335c043c5485e7e","observation_id":"16e9037b-f4fe-44d4-8fa8-a2fee29d1c6c","resolution":{"observed_at":"2026-07-04T16:39:58.343073Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2606.25658","last_updated":"2026-06-24T10:11:08Z","snapshot_observed_at":"2026-07-07T00:00:07.201776Z","submitted_at":"2026-06-24T10:11:08Z","title":"Towards a Dynamic and Fixed-budget Memory Bank for Efficient Streaming Video Understanding","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-06-25T20:54:49.319252Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2606.25658"},"observation_digest":"sha256:074a8b0db6424a73230685dbe027b9c483bcd4e298527f604c92592cf92b23bf","observation_id":"bf1fbb9e-037d-40dd-9ba6-a6f5309f6bd5","resolution":{"observed_at":"2026-07-04T20:00:08.184260Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-11T06:35:35.951554Z","title":"Flash-vstream: Memory- based real-time understanding for long video streams.arXiv preprint arXiv:2406.08085, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.05511","last_updated":"2026-07-06T18:00:06Z","snapshot_observed_at":"2026-08-04T18:10:17.416683Z","submitted_at":"2026-07-06T18:00:06Z","title":"Light-Omni: Reflex over Reasoning in Agentic Video Understanding with Long-Term Memory","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-07-11T06:35:35.951554Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2607.05511"},"observation_digest":"sha256:0b1c1288d5fc4e3f8c7f27c843bc94ff744bb1d33e110fe1530aa1375e2503a0","observation_id":"97a1906b-585e-4bfb-9a68-31c654f7fd22","resolution":{"observed_at":"2026-07-11T06:35:35.951554Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-14T05:01:06.200663Z","title":"arXiv preprint arXiv:2406.08085 (2024) 2, 4","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.11523","last_updated":"2026-07-13T13:09:45Z","snapshot_observed_at":"2026-07-16T23:19:51.232728Z","submitted_at":"2026-07-13T13:09:45Z","title":"Vinci2: Providing Proactive Assistance in Continuous Egocentric Videos","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-07-14T05:01:06.200663Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2607.11523"},"observation_digest":"sha256:89f59c77341797d389725335e885e1f4d7dd5f448f5839315d810e5d5566f0be","observation_id":"8226f807-a8ea-4980-a197-0d07ec3918a5","resolution":{"observed_at":"2026-07-14T05:01:06.200663Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-08-02T00:44:41.589936Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.14935","last_updated":"2026-07-16T12:47:59Z","snapshot_observed_at":"2026-08-02T14:24:21.558174Z","submitted_at":"2026-07-16T12:47:59Z","title":"VideoChat3: Fully Open Video MLLM for Efficient and Generalist Video Understanding","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-02T00:44:41.589936Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2607.14935"},"observation_digest":"sha256:5664340d5c6ae068df456233a6cfa401f65264714ec12a5c81115ff5af24e898","observation_id":"7e5440df-9915-40ca-8717-78f8973438bf","resolution":{"observed_at":"2026-08-02T00:44:41.589936Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-31T06:20:13.884368Z","title":"Flash-VStream: Memory-based real-time understanding for long video streams.arXiv preprint arXiv:2406.08085, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24904","last_updated":"2026-07-27T17:59:53Z","snapshot_observed_at":"2026-08-01T00:03:26.264908Z","submitted_at":"2026-07-27T17:59:53Z","title":"Mage-VL: An Efficient Codec-Native Streaming Multimodal Foundation Model","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-07-31T06:20:13.884368Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2607.24904"},"observation_digest":"sha256:70a2a47d0bd59b446c7d390b96be3d6c6664beb1a309ef20a4d951b4551e3936","observation_id":"d760ee98-5e90-420d-9bca-e53aecc7cd85","resolution":{"observed_at":"2026-07-31T06:20:13.884368Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-08-04T03:21:48.375332Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.28312","last_updated":"2026-08-01T16:00:11Z","snapshot_observed_at":"2026-08-05T22:16:21.572425Z","submitted_at":"2026-07-30T14:47:00Z","title":"ObjectStream: Latent Objects as Memory Anchors for Streaming Video Understanding","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-04T03:21:48.375332Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2607.28312"},"observation_digest":"sha256:b3e699ca688f3e3513ebce08e1fb5bc615cf5bd22e7adb8e929924380c8569e0","observation_id":"1b043da0-8aaa-49d4-99f7-8bb3ff0b33e3","resolution":{"observed_at":"2026-08-04T03:21:48.375332Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-08-03T00:45:21.931265Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.28678","last_updated":"2026-07-29T10:25:23Z","snapshot_observed_at":"2026-08-05T22:10:37.839636Z","submitted_at":"2026-07-29T10:25:23Z","title":"ViSAGE: Constructing Self-Correcting Memories for Long-Form Video Understanding","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-03T00:45:21.931265Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2607.28678"},"observation_digest":"sha256:f97148da1bc57759d71c47344667ff6f1e4ed21afdfe46d7b36c3fe906aa1384","observation_id":"9cea1237-dc36-4c46-8a46-6741e1a482ff","resolution":{"observed_at":"2026-08-03T00:45:21.931265Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-08-04T08:12:47.458517Z","title":"Zhang, X.; Jia, Z.; Guo, Z.; Li, J.; Li, B.; Li, H.; and Lu, Y","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.02392","last_updated":"2026-08-03T15:35:28Z","snapshot_observed_at":"2026-08-05T22:22:36.383495Z","submitted_at":"2026-08-03T15:35:28Z","title":"GROVE: Growing and Reasoning over Temporally Stratified Memory from Streaming Video Experience","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-04T08:12:47.458517Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2608.02392"},"observation_digest":"sha256:72b6b51adead476d8b2ac552501737feb6d7dc12d3945a7a29facd2ae8bd8315","observation_id":"259dbeea-6a1e-41a5-89ad-a27f7e1e4aaf","resolution":{"observed_at":"2026-08-04T08:12:47.458517Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2406.08085/citation-record","integrity":"/paper/2406.08085/integrity","json":"/paper/2406.08085/citation-record.json","paper":"/paper/2406.08085"},"outbound":[],"paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"thesis":"As of 5 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 43 inbound Pith citation observations for arXiv:2406.08085."}