{"as_of":"2026-08-10T03:18:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:e28943b5079a7f9328cf7e26c6f536f4b0393f3e0ec8bf6bbca0a97f69a7ef98","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":24,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":24,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":24,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":24,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:09:13.898584Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T03:19:29.878882Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2501.13468","last_updated":"2025-01-23T08:33:10Z","snapshot_observed_at":"2026-08-07T07:01:57.927733Z","submitted_at":"2025-01-23T08:33:10Z","title":"Streaming Video Understanding and Multi-round Interaction with Memory-enhanced Knowledge","version":1},"cited_work":{"arxiv_id":"2501.13468","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.13468","snapshot_observed_at":"2026-07-04T03:19:29.878882Z","title":"Streaming video under- standing and multi-round interaction with memory-enhanced knowledge","venue":null,"work_id":"05bede93-84f8-4b7b-b6e5-10f6ff02394b","year":2025},"citing_paper":{"arxiv_id":"2505.07062","last_updated":"2025-05-11T17:28:30Z","snapshot_observed_at":"2026-08-02T16:13:31.498470Z","submitted_at":"2025-05-11T17:28:30Z","title":"Seed1.5-VL Technical Report","version":1},"reference_index":154,"source":"pdf_text","source_observed_at":"2026-05-11T05:26:04.960844Z"},"links":{"cited_paper":"/paper/2501.13468","citing_paper":"/paper/2505.07062"},"observation_digest":"sha256:db67489bd23cc5f46ef072c32af2c071b8d0f91e68640db96afdbb651ae25016","observation_id":"1812f63e-756e-42b0-9447-5e044e886567","resolution":{"observed_at":"2026-05-11T05:26:05.736163Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13468","last_updated":"2025-01-23T08:33:10Z","snapshot_observed_at":"2026-08-07T07:01:57.927733Z","submitted_at":"2025-01-23T08:33:10Z","title":"Streaming Video Understanding and Multi-round Interaction with Memory-enhanced Knowledge","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13468","snapshot_observed_at":"2026-08-07T14:09:13.898584Z","title":"Streaming video understanding and multi-round interaction with memory-enhanced knowl- edge","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.19877","last_updated":"2025-05-26T12:05:16Z","snapshot_observed_at":"2026-08-09T16:51:01.852840Z","submitted_at":"2025-05-26T12:05:16Z","title":"Vad-R1: Towards Video Anomaly Reasoning via Perception-to-Cognition Chain-of-Thought","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-07T14:09:13.898584Z"},"links":{"cited_paper":"/paper/2501.13468","citing_paper":"/paper/2505.19877"},"observation_digest":"sha256:967cfa7e8a81404190652965b96851ac3f0808c16a6179d87b22890f36a5d135","observation_id":"e57d3da4-cce2-4184-93a0-e8ea7e0e0757","resolution":{"observed_at":"2026-08-07T14:09:13.898584Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13468","last_updated":"2025-01-23T08:33:10Z","snapshot_observed_at":"2026-08-07T07:01:57.927733Z","submitted_at":"2025-01-23T08:33:10Z","title":"Streaming Video Understanding and Multi-round Interaction with Memory-enhanced Knowledge","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13468","snapshot_observed_at":"2026-08-06T05:25:57.560559Z","title":"Preprint, arXiv:2501.13468","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.01874","last_updated":"2025-08-03T18:09:40Z","snapshot_observed_at":"2026-08-06T05:25:47.117242Z","submitted_at":"2025-08-03T18:09:40Z","title":"Diffractive electroproduction of light vector particles: leading Fock-state contribution in the presence of significant higher Fock-state effects","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T05:25:57.560559Z"},"links":{"cited_paper":"/paper/2501.13468","citing_paper":"/paper/2508.01874"},"observation_digest":"sha256:22004e1563abd8d7cd99b7598ff50a5e626dda81c587bc0ab62969f0ff05e62e","observation_id":"bec48dd9-f5e8-467c-bf47-6c9ff591f81e","resolution":{"observed_at":"2026-08-06T05:25:57.560559Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13468","last_updated":"2025-01-23T08:33:10Z","snapshot_observed_at":"2026-08-07T07:01:57.927733Z","submitted_at":"2025-01-23T08:33:10Z","title":"Streaming Video Understanding and Multi-round Interaction with Memory-enhanced Knowledge","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13468","snapshot_observed_at":"2026-08-04T20:20:36.908527Z","title":"Streaming video understanding and multi-round interaction with memory- enhanced knowledge.arXiv preprint arXiv:2501.13468,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.08621","last_updated":"2025-09-10T14:17:53Z","snapshot_observed_at":"2026-08-09T06:09:37.758584Z","submitted_at":"2025-09-10T14:17:53Z","title":"AdsQA: Towards Advertisement Video Understanding","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-04T20:20:36.908527Z"},"links":{"cited_paper":"/paper/2501.13468","citing_paper":"/paper/2509.08621"},"observation_digest":"sha256:bd87634ea02da00ef200db65a224123061f8e30e0faf2ba913e0fdc886e53dc1","observation_id":"ea305080-f6c1-45e1-b84d-4a875a2b6e31","resolution":{"observed_at":"2026-08-04T20:20:36.908527Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13468","last_updated":"2025-01-23T08:33:10Z","snapshot_observed_at":"2026-08-07T07:01:57.927733Z","submitted_at":"2025-01-23T08:33:10Z","title":"Streaming Video Understanding and Multi-round Interaction with Memory-enhanced Knowledge","version":1},"cited_work":{"arxiv_id":"2501.13468","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.13468","snapshot_observed_at":"2026-07-04T03:19:29.878882Z","title":"Streaming video under- standing and multi-round interaction with memory-enhanced knowledge","venue":null,"work_id":"05bede93-84f8-4b7b-b6e5-10f6ff02394b","year":2025},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"cited_paper":"/paper/2501.13468","citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:22bf7db302cbc74051d383683cded0647c5bb1504b5ad9dcc0c928a9d0172026","observation_id":"96708282-2c6f-4961-b694-275db5656e7e","resolution":{"observed_at":"2026-05-17T02:51:28.098757Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13468","last_updated":"2025-01-23T08:33:10Z","snapshot_observed_at":"2026-08-07T07:01:57.927733Z","submitted_at":"2025-01-23T08:33:10Z","title":"Streaming Video Understanding and Multi-round Interaction with Memory-enhanced Knowledge","version":1},"cited_work":{"arxiv_id":"2501.13468","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.13468","snapshot_observed_at":"2026-07-04T03:19:29.878882Z","title":"Streaming video under- standing and multi-round interaction with memory-enhanced knowledge","venue":null,"work_id":"05bede93-84f8-4b7b-b6e5-10f6ff02394b","year":2025},"citing_paper":{"arxiv_id":"2512.21334","last_updated":"2026-04-10T15:00:46Z","snapshot_observed_at":"2026-08-03T08:10:30.527963Z","submitted_at":"2025-12-24T18:59:36Z","title":"Streaming Video Instruction Tuning","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-16T19:44:11.032898Z"},"links":{"cited_paper":"/paper/2501.13468","citing_paper":"/paper/2512.21334"},"observation_digest":"sha256:1a635a4cb941e85afff8b977f5951cce3261a801f5a72fd64b2f37e1491a37ff","observation_id":"e95cc4b0-ed20-433f-a143-c9f4b56b3e17","resolution":{"observed_at":"2026-05-16T19:48:21.837952Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13468","last_updated":"2025-01-23T08:33:10Z","snapshot_observed_at":"2026-08-07T07:01:57.927733Z","submitted_at":"2025-01-23T08:33:10Z","title":"Streaming Video Understanding and Multi-round Interaction with Memory-enhanced Knowledge","version":1},"cited_work":{"arxiv_id":"2501.13468","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.13468","snapshot_observed_at":"2026-07-04T03:19:29.878882Z","title":"Streaming video under- standing and multi-round interaction with memory-enhanced knowledge","venue":null,"work_id":"05bede93-84f8-4b7b-b6e5-10f6ff02394b","year":2025},"citing_paper":{"arxiv_id":"2601.14724","last_updated":"2026-05-07T12:10:26Z","snapshot_observed_at":"2026-08-05T03:02:47.238051Z","submitted_at":"2026-01-21T07:26:15Z","title":"HERMES: KV Cache as Hierarchical Memory for Efficient Streaming Video Understanding","version":4},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-16T12:55:04.564442Z"},"links":{"cited_paper":"/paper/2501.13468","citing_paper":"/paper/2601.14724"},"observation_digest":"sha256:d4d605a12a832d0e74a8ff664809cd7a2d9559adc6a44bc5c6e581baaa6355fe","observation_id":"10b406f3-de90-4880-aaa4-951d4684bfcd","resolution":{"observed_at":"2026-05-16T12:57:53.814670Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13468","last_updated":"2025-01-23T08:33:10Z","snapshot_observed_at":"2026-08-07T07:01:57.927733Z","submitted_at":"2025-01-23T08:33:10Z","title":"Streaming Video Understanding and Multi-round Interaction with Memory-enhanced Knowledge","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13468","snapshot_observed_at":"2026-08-02T19:33:04.902520Z","title":"Streaming video understanding and multi-round interaction with memory-enhanced knowl- edge.arXiv preprint arXiv:2501.13468,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2603.01761","last_updated":"2026-06-16T11:36:48Z","snapshot_observed_at":"2026-08-08T18:03:36.079735Z","submitted_at":"2026-03-02T11:40:05Z","title":"Position: Modular Memory is the Key to Continual Learning Agents","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-02T19:33:04.902520Z"},"links":{"cited_paper":"/paper/2501.13468","citing_paper":"/paper/2603.01761"},"observation_digest":"sha256:cd97f769231945d84a23517f0d2963c82703d1706a729899e138e6c18f3bf18c","observation_id":"4640387a-ed4b-40e0-98ef-a9f34eedddce","resolution":{"observed_at":"2026-08-02T19:33:04.902520Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13468","last_updated":"2025-01-23T08:33:10Z","snapshot_observed_at":"2026-08-07T07:01:57.927733Z","submitted_at":"2025-01-23T08:33:10Z","title":"Streaming Video Understanding and Multi-round Interaction with Memory-enhanced Knowledge","version":1},"cited_work":{"arxiv_id":"2501.13468","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.13468","snapshot_observed_at":"2026-07-04T03:19:29.878882Z","title":"Streaming video under- standing and multi-round interaction with memory-enhanced knowledge","venue":null,"work_id":"05bede93-84f8-4b7b-b6e5-10f6ff02394b","year":2025},"citing_paper":{"arxiv_id":"2604.09000","last_updated":"2026-04-23T08:05:34Z","snapshot_observed_at":"2026-07-06T22:57:59.555972Z","submitted_at":"2026-04-10T06:11:34Z","title":"StreamMeCo: Long-Term Agent Memory Compression for Efficient Streaming Video Understanding","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-10T18:19:41.543165Z"},"links":{"cited_paper":"/paper/2501.13468","citing_paper":"/paper/2604.09000"},"observation_digest":"sha256:1e713990756b7346a94a13cbbbf1f9b48b2343810c0aed90cebce4065235bbb0","observation_id":"c2babc68-f2f8-44f2-af5c-bb9e26c79346","resolution":{"observed_at":"2026-05-11T00:45:50.440534Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13468","last_updated":"2025-01-23T08:33:10Z","snapshot_observed_at":"2026-08-07T07:01:57.927733Z","submitted_at":"2025-01-23T08:33:10Z","title":"Streaming Video Understanding and Multi-round Interaction with Memory-enhanced Knowledge","version":1},"cited_work":{"arxiv_id":"2501.13468","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.13468","snapshot_observed_at":"2026-07-04T03:19:29.878882Z","title":"Streaming video under- standing and multi-round interaction with memory-enhanced knowledge","venue":null,"work_id":"05bede93-84f8-4b7b-b6e5-10f6ff02394b","year":2025},"citing_paper":{"arxiv_id":"2604.11411","last_updated":"2026-04-13T12:55:56Z","snapshot_observed_at":"2026-08-03T03:42:10.089825Z","submitted_at":"2026-04-13T12:55:56Z","title":"Online Reasoning Video Object Segmentation","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-10T15:29:03.843441Z"},"links":{"cited_paper":"/paper/2501.13468","citing_paper":"/paper/2604.11411"},"observation_digest":"sha256:1b3f9649850a40ca1c4474a1e1805b8a1891768c33664f17c50371bea64f261a","observation_id":"9ee79b02-aca4-4767-8d12-6b0124b5ed74","resolution":{"observed_at":"2026-05-11T10:26:02.856020Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13468","last_updated":"2025-01-23T08:33:10Z","snapshot_observed_at":"2026-08-07T07:01:57.927733Z","submitted_at":"2025-01-23T08:33:10Z","title":"Streaming Video Understanding and Multi-round Interaction with Memory-enhanced Knowledge","version":1},"cited_work":{"arxiv_id":"2501.13468","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.13468","snapshot_observed_at":"2026-07-04T03:19:29.878882Z","title":"Streaming video under- standing and multi-round interaction with memory-enhanced knowledge","venue":null,"work_id":"05bede93-84f8-4b7b-b6e5-10f6ff02394b","year":2025},"citing_paper":{"arxiv_id":"2604.17052","last_updated":"2026-04-18T16:22:05Z","snapshot_observed_at":"2026-08-08T21:49:06.109128Z","submitted_at":"2026-04-18T16:22:05Z","title":"OASIS: On-Demand Hierarchical Event Memory for Streaming Video Reasoning","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-10T06:51:52.861981Z"},"links":{"cited_paper":"/paper/2501.13468","citing_paper":"/paper/2604.17052"},"observation_digest":"sha256:dd8bb47e6bdba989771141d4517abe44b66358c8294e9a691a762aa157490082","observation_id":"c1dcbb80-6dad-4874-a696-0d5d79bdd3e5","resolution":{"observed_at":"2026-05-10T06:56:47.801677Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13468","last_updated":"2025-01-23T08:33:10Z","snapshot_observed_at":"2026-08-07T07:01:57.927733Z","submitted_at":"2025-01-23T08:33:10Z","title":"Streaming Video Understanding and Multi-round Interaction with Memory-enhanced Knowledge","version":1},"cited_work":{"arxiv_id":"2501.13468","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.13468","snapshot_observed_at":"2026-07-04T03:19:29.878882Z","title":"Streaming video under- standing and multi-round interaction with memory-enhanced knowledge","venue":null,"work_id":"05bede93-84f8-4b7b-b6e5-10f6ff02394b","year":2025},"citing_paper":{"arxiv_id":"2604.24317","last_updated":"2026-04-27T11:07:03Z","snapshot_observed_at":"2026-07-06T23:10:24.992210Z","submitted_at":"2026-04-27T11:07:03Z","title":"Don't Pause! Every prediction matters in a streaming video","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-05-08T04:32:01.379605Z"},"links":{"cited_paper":"/paper/2501.13468","citing_paper":"/paper/2604.24317"},"observation_digest":"sha256:d604af79871ae588682ada72626b29270c9c6d286a9102f634bccec533c4017f","observation_id":"7430da5b-7941-4ebb-849a-1d994f51f855","resolution":{"observed_at":"2026-05-11T21:41:18.108827Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13468","last_updated":"2025-01-23T08:33:10Z","snapshot_observed_at":"2026-08-07T07:01:57.927733Z","submitted_at":"2025-01-23T08:33:10Z","title":"Streaming Video Understanding and Multi-round Interaction with Memory-enhanced Knowledge","version":1},"cited_work":{"arxiv_id":"2501.13468","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.13468","snapshot_observed_at":"2026-07-04T03:19:29.878882Z","title":"Streaming video under- standing and multi-round interaction with memory-enhanced knowledge","venue":null,"work_id":"05bede93-84f8-4b7b-b6e5-10f6ff02394b","year":2025},"citing_paper":{"arxiv_id":"2605.07575","last_updated":"2026-05-11T11:58:53Z","snapshot_observed_at":"2026-07-06T23:19:56.415523Z","submitted_at":"2026-05-08T10:46:10Z","title":"Response-G1: Explicit Scene Graph Modeling for Proactive Streaming Video Understanding","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-05-11T02:30:55.939351Z"},"links":{"cited_paper":"/paper/2501.13468","citing_paper":"/paper/2605.07575"},"observation_digest":"sha256:8247f08bd6973dbed17c54ce671f2af9beaea762a8b563bdb7cdbc05e8e8edb1","observation_id":"83bcec37-707f-44f6-8587-a9687b7594e2","resolution":{"observed_at":"2026-05-11T03:20:56.376815Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13468","last_updated":"2025-01-23T08:33:10Z","snapshot_observed_at":"2026-08-07T07:01:57.927733Z","submitted_at":"2025-01-23T08:33:10Z","title":"Streaming Video Understanding and Multi-round Interaction with Memory-enhanced Knowledge","version":1},"cited_work":{"arxiv_id":"2501.13468","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.13468","snapshot_observed_at":"2026-07-04T03:19:29.878882Z","title":"Streaming video under- standing and multi-round interaction with memory-enhanced knowledge","venue":null,"work_id":"05bede93-84f8-4b7b-b6e5-10f6ff02394b","year":2025},"citing_paper":{"arxiv_id":"2605.07575","last_updated":"2026-05-11T11:58:53Z","snapshot_observed_at":"2026-07-06T23:19:56.415523Z","submitted_at":"2026-05-08T10:46:10Z","title":"Response-G1: Explicit Scene Graph Modeling for Proactive Streaming Video Understanding","version":2},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-05-12T03:00:34.728880Z"},"links":{"cited_paper":"/paper/2501.13468","citing_paper":"/paper/2605.07575"},"observation_digest":"sha256:5a4ca18fe290e5a93c1070743cc79f72a99191bc6b06ad69b53dfb2da4f1a8e2","observation_id":"e68a8438-83f1-4109-b5cd-36c9b9750cb4","resolution":{"observed_at":"2026-05-12T03:01:17.745300Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13468","last_updated":"2025-01-23T08:33:10Z","snapshot_observed_at":"2026-08-07T07:01:57.927733Z","submitted_at":"2025-01-23T08:33:10Z","title":"Streaming Video Understanding and Multi-round Interaction with Memory-enhanced Knowledge","version":1},"cited_work":{"arxiv_id":"2501.13468","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.13468","snapshot_observed_at":"2026-07-04T03:19:29.878882Z","title":"Streaming video under- standing and multi-round interaction with memory-enhanced knowledge","venue":null,"work_id":"05bede93-84f8-4b7b-b6e5-10f6ff02394b","year":2025},"citing_paper":{"arxiv_id":"2605.25621","last_updated":"2026-05-25T09:23:19Z","snapshot_observed_at":"2026-07-06T23:35:36.158984Z","submitted_at":"2026-05-25T09:23:19Z","title":"StreamOV: Streaming Omni-Video Understanding via Evidence-Guided Memory and Response Triggering","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-06-29T22:27:17.092553Z"},"links":{"cited_paper":"/paper/2501.13468","citing_paper":"/paper/2605.25621"},"observation_digest":"sha256:d0e3f346b7ee0ba4352730ef8fa8de42d3a9882e599c22a7517d29c0e3642ad7","observation_id":"28611be7-b521-4a79-b59f-77353832f72c","resolution":{"observed_at":"2026-06-29T22:34:02.045071Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13468","last_updated":"2025-01-23T08:33:10Z","snapshot_observed_at":"2026-08-07T07:01:57.927733Z","submitted_at":"2025-01-23T08:33:10Z","title":"Streaming Video Understanding and Multi-round Interaction with Memory-enhanced Knowledge","version":1},"cited_work":{"arxiv_id":"2501.13468","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.13468","snapshot_observed_at":"2026-07-04T03:19:29.878882Z","title":"Streaming video under- standing and multi-round interaction with memory-enhanced knowledge","venue":null,"work_id":"05bede93-84f8-4b7b-b6e5-10f6ff02394b","year":2025},"citing_paper":{"arxiv_id":"2605.27074","last_updated":"2026-05-26T14:23:25Z","snapshot_observed_at":"2026-08-02T20:44:04.329111Z","submitted_at":"2026-05-26T14:23:25Z","title":"IPIBench: Evaluating Interactive Proactive Intelligence of MLLMs under Continuous Streams","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-06-29T18:24:57.881644Z"},"links":{"cited_paper":"/paper/2501.13468","citing_paper":"/paper/2605.27074"},"observation_digest":"sha256:28a48423acd8716df261a8bdef2ec00d53a9bde6a57268905f23bf6f7a5bfb00","observation_id":"aa4d9b63-b76a-4ffe-8f0d-c3c75a2c8f0d","resolution":{"observed_at":"2026-06-29T18:33:51.046724Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13468","last_updated":"2025-01-23T08:33:10Z","snapshot_observed_at":"2026-08-07T07:01:57.927733Z","submitted_at":"2025-01-23T08:33:10Z","title":"Streaming Video Understanding and Multi-round Interaction with Memory-enhanced Knowledge","version":1},"cited_work":{"arxiv_id":"2501.13468","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.13468","snapshot_observed_at":"2026-07-04T03:19:29.878882Z","title":"Streaming video under- standing and multi-round interaction with memory-enhanced knowledge","venue":null,"work_id":"05bede93-84f8-4b7b-b6e5-10f6ff02394b","year":2025},"citing_paper":{"arxiv_id":"2605.27318","last_updated":"2026-07-06T17:18:57Z","snapshot_observed_at":"2026-07-12T15:54:07.137141Z","submitted_at":"2026-05-26T17:26:29Z","title":"Q-GeoMem: Question-Guided Geometric Memory for Video Spatial Reasoning","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-06-29T18:15:28.261185Z"},"links":{"cited_paper":"/paper/2501.13468","citing_paper":"/paper/2605.27318"},"observation_digest":"sha256:7b410745119dfe43c4e24cbb576efe0d861a778840dfd92fb4c879cfafa81eac","observation_id":"8c46cb1b-9174-40ba-9407-3e7327970d1a","resolution":{"observed_at":"2026-06-29T18:23:50.977490Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13468","last_updated":"2025-01-23T08:33:10Z","snapshot_observed_at":"2026-08-07T07:01:57.927733Z","submitted_at":"2025-01-23T08:33:10Z","title":"Streaming Video Understanding and Multi-round Interaction with Memory-enhanced Knowledge","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13468","snapshot_observed_at":"2026-07-12T15:54:12.909974Z","title":"Streaming video understanding and multi-round interaction with memory-enhanced knowledge","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.27318","last_updated":"2026-07-06T17:18:57Z","snapshot_observed_at":"2026-07-12T15:54:07.137141Z","submitted_at":"2026-05-26T17:26:29Z","title":"Q-GeoMem: Question-Guided Geometric Memory for Video Spatial Reasoning","version":2},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-07-12T15:54:12.909974Z"},"links":{"cited_paper":"/paper/2501.13468","citing_paper":"/paper/2605.27318"},"observation_digest":"sha256:e6c1bea6267169ff29d49511d278e589e9ce48c338850578003453922ea7168e","observation_id":"568956ce-6d85-4e03-9139-ae656641ebf4","resolution":{"observed_at":"2026-07-12T15:54:12.909974Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13468","last_updated":"2025-01-23T08:33:10Z","snapshot_observed_at":"2026-08-07T07:01:57.927733Z","submitted_at":"2025-01-23T08:33:10Z","title":"Streaming Video Understanding and Multi-round Interaction with Memory-enhanced Knowledge","version":1},"cited_work":{"arxiv_id":"2501.13468","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.13468","snapshot_observed_at":"2026-07-04T03:19:29.878882Z","title":"Streaming video under- standing and multi-round interaction with memory-enhanced knowledge","venue":null,"work_id":"05bede93-84f8-4b7b-b6e5-10f6ff02394b","year":2025},"citing_paper":{"arxiv_id":"2605.31557","last_updated":"2026-06-01T11:50:06Z","snapshot_observed_at":"2026-07-06T23:40:42.277618Z","submitted_at":"2026-05-29T17:20:10Z","title":"EGOSTREAM: A Diagnostic Benchmark for Streaming Episodic Memory in Egocentric Vision","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-06-28T22:40:13.720341Z"},"links":{"cited_paper":"/paper/2501.13468","citing_paper":"/paper/2605.31557"},"observation_digest":"sha256:78f1a90ca55ead8d8dbf1e6a3634e9f97c68d40a22ec0764d28e4bc487772a5d","observation_id":"fc8ede71-d296-4845-9683-09091ca4b6a9","resolution":{"observed_at":"2026-06-28T22:42:46.375881Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13468","last_updated":"2025-01-23T08:33:10Z","snapshot_observed_at":"2026-08-07T07:01:57.927733Z","submitted_at":"2025-01-23T08:33:10Z","title":"Streaming Video Understanding and Multi-round Interaction with Memory-enhanced Knowledge","version":1},"cited_work":{"arxiv_id":"2501.13468","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.13468","snapshot_observed_at":"2026-07-04T03:19:29.878882Z","title":"Streaming video under- standing and multi-round interaction with memory-enhanced knowledge","venue":null,"work_id":"05bede93-84f8-4b7b-b6e5-10f6ff02394b","year":2025},"citing_paper":{"arxiv_id":"2606.05008","last_updated":"2026-06-03T15:28:57Z","snapshot_observed_at":"2026-08-06T07:23:12.183499Z","submitted_at":"2026-06-03T15:28:57Z","title":"M$^3$Eval: Multi-Modal Memory Evaluation through Cognitively-Grounded Video Tasks","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-06-28T06:16:07.090870Z"},"links":{"cited_paper":"/paper/2501.13468","citing_paper":"/paper/2606.05008"},"observation_digest":"sha256:e0a0d70d71cc53762efa8164a63f22cce25fd6477fb4ab8c3070018d271a2cb9","observation_id":"b6fdec21-3bea-44d9-bb3a-408d96c68408","resolution":{"observed_at":"2026-07-02T08:16:47.799122Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13468","last_updated":"2025-01-23T08:33:10Z","snapshot_observed_at":"2026-08-07T07:01:57.927733Z","submitted_at":"2025-01-23T08:33:10Z","title":"Streaming Video Understanding and Multi-round Interaction with Memory-enhanced Knowledge","version":1},"cited_work":{"arxiv_id":"2501.13468","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.13468","snapshot_observed_at":"2026-07-04T03:19:29.878882Z","title":"Streaming video under- standing and multi-round interaction with memory-enhanced knowledge","venue":null,"work_id":"05bede93-84f8-4b7b-b6e5-10f6ff02394b","year":2025},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"cited_paper":"/paper/2501.13468","citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:f26350d31d8c3d2d5a0dd51a1bb7e6b74ca8e07a7de6079dd756d99705c24acd","observation_id":"cc8386c5-477b-468e-8966-e97c33317c4c","resolution":{"observed_at":"2026-07-02T17:07:12.826651Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13468","last_updated":"2025-01-23T08:33:10Z","snapshot_observed_at":"2026-08-07T07:01:57.927733Z","submitted_at":"2025-01-23T08:33:10Z","title":"Streaming Video Understanding and Multi-round Interaction with Memory-enhanced Knowledge","version":1},"cited_work":{"arxiv_id":"2501.13468","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.13468","snapshot_observed_at":"2026-07-04T03:19:29.878882Z","title":"Streaming video under- standing and multi-round interaction with memory-enhanced knowledge","venue":null,"work_id":"05bede93-84f8-4b7b-b6e5-10f6ff02394b","year":2025},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-08-08T11:16:26.006355Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/2501.13468","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:6d198551a7d4019d5f0a38ddd405d260a4f9af9820b3fedb38ef4ecb633a845e","observation_id":"a9e49ad1-317e-4670-b9c3-23633883a83e","resolution":{"observed_at":"2026-07-03T20:38:56.211274Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13468","last_updated":"2025-01-23T08:33:10Z","snapshot_observed_at":"2026-08-07T07:01:57.927733Z","submitted_at":"2025-01-23T08:33:10Z","title":"Streaming Video Understanding and Multi-round Interaction with Memory-enhanced Knowledge","version":1},"cited_work":{"arxiv_id":"2501.13468","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.13468","snapshot_observed_at":"2026-07-04T03:19:29.878882Z","title":"Streaming video under- standing and multi-round interaction with memory-enhanced knowledge","venue":null,"work_id":"05bede93-84f8-4b7b-b6e5-10f6ff02394b","year":2025},"citing_paper":{"arxiv_id":"2606.19849","last_updated":"2026-06-18T06:57:31Z","snapshot_observed_at":"2026-08-06T15:40:32.013079Z","submitted_at":"2026-06-18T06:57:31Z","title":"ViCoStream: Streaming VideoLLMs Can Run Beyond 100 FPS with Stage-Wise Coordinated Inference","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-06-26T18:17:53.013043Z"},"links":{"cited_paper":"/paper/2501.13468","citing_paper":"/paper/2606.19849"},"observation_digest":"sha256:ec315fa2e7c416153ace82593f00b7414bdfcebbadbec1a4028259391e056d71","observation_id":"ddcad86d-fe93-4418-b99d-a8e7fa07a6d4","resolution":{"observed_at":"2026-07-04T03:19:29.882474Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13468","last_updated":"2025-01-23T08:33:10Z","snapshot_observed_at":"2026-08-07T07:01:57.927733Z","submitted_at":"2025-01-23T08:33:10Z","title":"Streaming Video Understanding and Multi-round Interaction with Memory-enhanced Knowledge","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13468","snapshot_observed_at":"2026-08-02T05:40:48.131445Z","title":"Streaming video under- standing and multi-round interaction with memory-enhanced knowledge.arXiv preprint arXiv:2501.13468, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.13298","last_updated":"2026-07-14T22:09:48Z","snapshot_observed_at":"2026-08-07T01:40:11.166031Z","submitted_at":"2026-07-14T22:09:48Z","title":"FOLIO: Focused Semantic Memory for Streaming Video Understanding","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-02T05:40:48.131445Z"},"links":{"cited_paper":"/paper/2501.13468","citing_paper":"/paper/2607.13298"},"observation_digest":"sha256:1f498f8ac826cf9e8580d1175ec6b583a75cf53b337fdc3fcfd813f63d7591e7","observation_id":"a35e0a5b-0773-466f-8269-8709cf983134","resolution":{"observed_at":"2026-08-02T05:40:48.131445Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2501.13468/citation-record","integrity":"/paper/2501.13468/integrity","json":"/paper/2501.13468/citation-record.json","paper":"/paper/2501.13468"},"outbound":[],"paper":{"arxiv_id":"2501.13468","last_updated":"2025-01-23T08:33:10Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-07T07:01:57.927733Z","submitted_at":"2025-01-23T08:33:10Z","title":"Streaming Video Understanding and Multi-round Interaction with Memory-enhanced Knowledge"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 24 inbound Pith citation observations for arXiv:2501.13468."}