{"as_of":"2026-08-02T23:03:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:f780aaf457f47ef3f9d0fbebd41506a890a4babaa1a3027b83f6f6842c58764b","coverage":[{"denominator":19,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":19,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-10T14:47:54.917408Z","state":"measured"},{"denominator":20,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":20,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-02T06:30:47.504484+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-01T12:28:00.885137Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2605.01858","last_updated":"2026-05-03T13:02:44Z","snapshot_observed_at":"2026-07-06T23:15:01.968194Z","submitted_at":"2026-05-03T13:02:44Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.01858","snapshot_observed_at":"2026-08-01T12:28:00.885137Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.19547","last_updated":"2026-07-21T19:46:38Z","snapshot_observed_at":"2026-08-01T12:28:00.037667Z","submitted_at":"2026-07-21T19:46:38Z","title":"ChronoStitch: Training-Free Composition of Visual KV Memories for Long-Horizon Temporal Reasoning","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-01T12:28:00.885137Z"},"links":{"cited_paper":"/paper/2605.01858","citing_paper":"/paper/2607.19547"},"observation_digest":"sha256:8cdbff7fa8dc127a7c5ea0ad02d46f4ff1658749287f35d95543507afae7c9a5","observation_id":"9a879610-a4d1-493d-9ec5-98b5baa78904","resolution":{"observed_at":"2026-08-01T12:28:00.885137Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2605.01858/citation-record","integrity":"/paper/2605.01858/integrity","json":"/paper/2605.01858/citation-record.json","paper":"/paper/2605.01858"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2502.13923","last_updated":"2025-02-19T18:00:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-19T18:00:14Z","title":"Qwen2.5-VL Technical Report","version":1},"cited_work":{"arxiv_id":"2502.13923","doi":"10.48550/arxiv.2502.13923","metadata_source":"pith","pith_arxiv_id":"2502.13923","snapshot_observed_at":"2026-07-11T01:17:45.013705Z","title":"Qwen2.5-VL Technical Report","venue":"cs.CV","work_id":"69dffacb-bfe8-442d-be86-48624c60426f","year":2025},"citing_paper":{"arxiv_id":"2605.01858","last_updated":"2026-05-03T13:02:44Z","snapshot_observed_at":"2026-07-06T23:15:01.968194Z","submitted_at":"2026-05-03T13:02:44Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T14:47:54.917408Z"},"links":{"cited_paper":"/paper/2502.13923","citing_paper":"/paper/2605.01858"},"observation_digest":"sha256:9b1a54f56442b79f2fef7abc246189df87a4feeb2a1f5738136136d125dffa48","observation_id":"27b29a5d-81b8-479e-8e4a-c7c3431e3540","resolution":{"observed_at":"2026-05-11T11:31:03.544517Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:19:13.082554+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:19:13.082554+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.13915","last_updated":"2025-04-10T17:13:08Z","snapshot_observed_at":"2026-07-06T21:11:39.503336Z","submitted_at":"2025-04-10T17:13:08Z","title":"Memory-efficient Streaming VideoLLMs for Real-time Procedural Video Understanding","version":1},"cited_work":{"arxiv_id":"2504.13915","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2504.13915","snapshot_observed_at":"2026-07-04T13:19:50.790416Z","title":"Memory-efficient streaming videollms for real-time procedural video understanding","venue":null,"work_id":"74bc44fa-7cf0-445e-a837-49c5d4faab49","year":2025},"citing_paper":{"arxiv_id":"2605.01858","last_updated":"2026-05-03T13:02:44Z","snapshot_observed_at":"2026-07-06T23:15:01.968194Z","submitted_at":"2026-05-03T13:02:44Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-10T14:47:54.917408Z"},"links":{"cited_paper":"/paper/2504.13915","citing_paper":"/paper/2605.01858"},"observation_digest":"sha256:fc0b87b56fd2849d6dda9c17cb19fad07fec1579d3663463bfdacaea31745a1a","observation_id":"aae0a56c-695d-47e3-9cfc-69be4f57db2e","resolution":{"observed_at":"2026-05-11T11:31:03.596964Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.15595","last_updated":"2023-06-28T04:26:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-27T16:26:26Z","title":"Extending Context Window of Large Language Models via Positional Interpolation","version":2},"cited_work":{"arxiv_id":"2306.15595","doi":"10.48550/arxiv.2306.15595","metadata_source":"pith","pith_arxiv_id":"2306.15595","snapshot_observed_at":"2026-07-10T20:07:33.466282Z","title":"Extending Context Window of Large Language Models via Positional Interpolation","venue":"cs.CL","work_id":"c8b6df85-e7da-4bd8-90a4-d309cc2a0f60","year":2023},"citing_paper":{"arxiv_id":"2605.01858","last_updated":"2026-05-03T13:02:44Z","snapshot_observed_at":"2026-07-06T23:15:01.968194Z","submitted_at":"2026-05-03T13:02:44Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-10T14:47:54.917408Z"},"links":{"cited_paper":"/paper/2306.15595","citing_paper":"/paper/2605.01858"},"observation_digest":"sha256:27599ca0a44f168d668c2b8a4ee7e06096d0ba705d333f975e033866785a13e0","observation_id":"bb0993fe-c244-4af7-ab40-485b7c61345d","resolution":{"observed_at":"2026-05-13T10:20:58.016961Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-01T10:38:12.283235+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T10:38:12.283235+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.12769","last_updated":"2025-03-17T03:05:31Z","snapshot_observed_at":"2026-07-06T20:53:41.124209Z","submitted_at":"2025-03-17T03:05:31Z","title":"ViSpeak: Visual Instruction Feedback in Streaming Videos","version":1},"cited_work":{"arxiv_id":"2503.12769","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.12769","snapshot_observed_at":"2026-07-02T19:07:17.481649Z","title":"Vispeak: Visual instruction feedback in streaming videos","venue":null,"work_id":"ff02f96e-6220-439c-a2d7-ebe7d4b79876","year":2025},"citing_paper":{"arxiv_id":"2605.01858","last_updated":"2026-05-03T13:02:44Z","snapshot_observed_at":"2026-07-06T23:15:01.968194Z","submitted_at":"2026-05-03T13:02:44Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-10T14:47:54.917408Z"},"links":{"cited_paper":"/paper/2503.12769","citing_paper":"/paper/2605.01858"},"observation_digest":"sha256:c115f33ae971da7acdc9569fa28de74c15039b9cb1ad90c4f7923941c3313eba","observation_id":"c43ba94c-104b-4986-8759-33939e277c70","resolution":{"observed_at":"2026-05-11T11:31:03.582625Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Lm-infinite: Zero-shot extreme length generalization for large language models","venue":null,"work_id":"9a6d1e54-beca-4048-a619-ed17c8d66fe2","year":2024},"citing_paper":{"arxiv_id":"2605.01858","last_updated":"2026-05-03T13:02:44Z","snapshot_observed_at":"2026-07-06T23:15:01.968194Z","submitted_at":"2026-05-03T13:02:44Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-10T14:47:54.917408Z"},"links":{"citing_paper":"/paper/2605.01858"},"observation_digest":"sha256:9464173731df7426900a6b1fc142b25989a97aa84f517bb99afd803caef9621f","observation_id":"366c7205-f8de-4192-b100-b535d2072998","resolution":{"observed_at":"2026-05-18T11:46:19.677859Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2506.15745","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T16:39:58.347155Z","title":"Infinipot-v: Memory-constrained kv cache com- pression for streaming video understanding","venue":null,"work_id":"798f7dbf-55e6-44c0-9508-6aec9da8c639","year":2025},"citing_paper":{"arxiv_id":"2605.01858","last_updated":"2026-05-03T13:02:44Z","snapshot_observed_at":"2026-07-06T23:15:01.968194Z","submitted_at":"2026-05-03T13:02:44Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-10T14:47:54.917408Z"},"links":{"citing_paper":"/paper/2605.01858"},"observation_digest":"sha256:74727828441cb4c1bd5f43a194ad3b9bf1f5410a58c1138d98f7b7bbeb2f339c","observation_id":"912e9849-5a85-46c4-ada0-1164ec692f17","resolution":{"observed_at":"2026-05-11T11:31:03.592378Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03326","last_updated":"2024-10-26T16:35:13Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-06T17:59:44Z","title":"LLaVA-OneVision: Easy Visual Task Transfer","version":3},"cited_work":{"arxiv_id":"2408.03326","doi":"10.48550/arxiv.2408.03326","metadata_source":"pith","pith_arxiv_id":"2408.03326","snapshot_observed_at":"2026-07-10T13:27:05.574543Z","title":"LLaVA-OneVision: Easy Visual Task Transfer","venue":"cs.CV","work_id":"f5f2452b-f2a9-49ac-b38d-c76e18cdfe49","year":2024},"citing_paper":{"arxiv_id":"2605.01858","last_updated":"2026-05-03T13:02:44Z","snapshot_observed_at":"2026-07-06T23:15:01.968194Z","submitted_at":"2026-05-03T13:02:44Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-10T14:47:54.917408Z"},"links":{"cited_paper":"/paper/2408.03326","citing_paper":"/paper/2605.01858"},"observation_digest":"sha256:94af5c3380e62b9c959d15b3616a3da922182b0b46539432723b2d56ddc93796","observation_id":"f52bea55-7459-4551-a43f-d5c408a2e669","resolution":{"observed_at":"2026-05-11T11:31:03.558420Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-07-06T19:45:57.808548Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":"2411.03628","doi":"10.48550/arxiv.2411.03628","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Streamingbench: Assessing the gap for mllms to achieve streaming video un- derstanding","venue":null,"work_id":"caec985b-dd2d-4dbb-9199-d147732de99a","year":2024},"citing_paper":{"arxiv_id":"2605.01858","last_updated":"2026-05-03T13:02:44Z","snapshot_observed_at":"2026-07-06T23:15:01.968194Z","submitted_at":"2026-05-03T13:02:44Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-10T14:47:54.917408Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2605.01858"},"observation_digest":"sha256:feb2b0db9660fb5355506f880b8da01ce7943e8bd673f1360537affd69e22323","observation_id":"eb3684ab-300c-44e6-a7d0-7b730bf49737","resolution":{"observed_at":"2026-05-11T11:31:03.575498Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.03172","last_updated":"2023-11-20T23:09:34Z","snapshot_observed_at":"2026-07-06T15:51:12.179086Z","submitted_at":"2023-07-06T17:54:11Z","title":"Lost in the Middle: How Language Models Use Long Contexts","version":3},"cited_work":{"arxiv_id":"2307.03172","doi":"10.1162/tacl","metadata_source":"pith","pith_arxiv_id":"2307.03172","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Lost in the Middle: How Language Models Use Long Contexts","venue":"cs.CL","work_id":"37c05e13-4a24-44f8-a1c4-da1bbe7223aa","year":2023},"citing_paper":{"arxiv_id":"2605.01858","last_updated":"2026-05-03T13:02:44Z","snapshot_observed_at":"2026-07-06T23:15:01.968194Z","submitted_at":"2026-05-03T13:02:44Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-10T14:47:54.917408Z"},"links":{"cited_paper":"/paper/2307.03172","citing_paper":"/paper/2605.01858"},"observation_digest":"sha256:029436a83877e69f46bf758644ec52992e598a8b9e93ff84ccac9b4576ba4895","observation_id":"f391ef84-cbb1-40bb-bf2f-1326a2d7e83a","resolution":{"observed_at":"2026-05-11T11:31:03.570719Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15269","last_updated":"2026-04-23T12:54:38Z","snapshot_observed_at":"2026-07-06T21:27:42.002746Z","submitted_at":"2025-05-21T08:47:15Z","title":"LiveVLM: Efficient Online Video Understanding via Streaming-Oriented KV Cache and Retrieval","version":2},"cited_work":{"arxiv_id":"2505.15269","doi":"10.48550/arxiv.2505.15269","metadata_source":"pith","pith_arxiv_id":"2505.15269","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"LiveVLM: Efficient Online Video Understanding via Streaming-Oriented KV Cache and Retrieval","venue":"cs.CV","work_id":"a00b4d7a-5adc-4855-89a1-5356dc946a85","year":2025},"citing_paper":{"arxiv_id":"2605.01858","last_updated":"2026-05-03T13:02:44Z","snapshot_observed_at":"2026-07-06T23:15:01.968194Z","submitted_at":"2026-05-03T13:02:44Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-10T14:47:54.917408Z"},"links":{"cited_paper":"/paper/2505.15269","citing_paper":"/paper/2605.01858"},"observation_digest":"sha256:43c3d2cf91c1ff390874f0c45b03ecb34dea626cdd947c7418e0a5499fd6f9b9","observation_id":"813374e9-2614-4a74-b24e-6d7c58b750b5","resolution":{"observed_at":"2026-05-11T11:31:03.587336Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2603.02546","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"On discrim- inative vs","venue":null,"work_id":"e6049f41-8edb-46d7-a337-6b57a271c937","year":null},"citing_paper":{"arxiv_id":"2605.01858","last_updated":"2026-05-03T13:02:44Z","snapshot_observed_at":"2026-07-06T23:15:01.968194Z","submitted_at":"2026-05-03T13:02:44Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-10T14:47:54.917408Z"},"links":{"citing_paper":"/paper/2605.01858"},"observation_digest":"sha256:1e9bf654b9445aab01f78d5f1b6c36d7ba2fbf7e9c804205491ff52cef8c731a","observation_id":"6ce05c96-c77c-4e1f-be56-354d722a8297","resolution":{"observed_at":"2026-05-11T11:31:03.612285Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2108.12409","last_updated":"2022-04-22T18:20:48Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-08-27T17:35:06Z","title":"Train Short, Test Long: Attention with Linear Biases Enables Input Length Extrapolation","version":2},"cited_work":{"arxiv_id":"2108.12409","doi":"10.48550/arxiv.2108.12409","metadata_source":"pith","pith_arxiv_id":"2108.12409","snapshot_observed_at":"2026-07-10T20:07:33.543353Z","title":"Train Short, Test Long: Attention with Linear Biases Enables Input Length Extrapolation","venue":"cs.CL","work_id":"145b1374-5258-4c00-a433-4db0f5a50749","year":2021},"citing_paper":{"arxiv_id":"2605.01858","last_updated":"2026-05-03T13:02:44Z","snapshot_observed_at":"2026-07-06T23:15:01.968194Z","submitted_at":"2026-05-03T13:02:44Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-10T14:47:54.917408Z"},"links":{"cited_paper":"/paper/2108.12409","citing_paper":"/paper/2605.01858"},"observation_digest":"sha256:4c5cc710c33461962f70338ac07932a6673e20d0f66f5c59d9dfb97f488dfe75","observation_id":"c3db2087-ef04-4486-b0e0-84246ad47a2e","resolution":{"observed_at":"2026-05-13T01:01:23.440331Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05530","last_updated":"2024-12-16T17:39:39Z","snapshot_observed_at":"2026-07-06T17:41:42.995949Z","submitted_at":"2024-03-08T18:54:20Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","version":5},"cited_work":{"arxiv_id":"2403.05530","doi":"10.48550/arxiv.2403.05530","metadata_source":"pith","pith_arxiv_id":"2403.05530","snapshot_observed_at":"2026-07-11T03:37:46.178537Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","venue":"cs.CL","work_id":"80e3e977-f1bb-4c83-8d0c-1ab0a0c5c3f1","year":2024},"citing_paper":{"arxiv_id":"2605.01858","last_updated":"2026-05-03T13:02:44Z","snapshot_observed_at":"2026-07-06T23:15:01.968194Z","submitted_at":"2026-05-03T13:02:44Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-10T14:47:54.917408Z"},"links":{"cited_paper":"/paper/2403.05530","citing_paper":"/paper/2605.01858"},"observation_digest":"sha256:cce63923be76b79a39d2ad748910344cd8eaaaa00486e2b8f089930e6bb866bc","observation_id":"9e0f6efa-4005-45a8-85ac-57d3c18dbf16","resolution":{"observed_at":"2026-05-11T11:31:03.649723Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T03:08:14.426583+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T03:08:14.426583+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2505.05467","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-02T17:27:15.523830Z","title":"Streambridge: Turning your offline video large language model into a proactive streaming assistant","venue":null,"work_id":"6f0d7f4a-5862-4b5f-8e70-eeaff1ed8a54","year":2025},"citing_paper":{"arxiv_id":"2605.01858","last_updated":"2026-05-03T13:02:44Z","snapshot_observed_at":"2026-07-06T23:15:01.968194Z","submitted_at":"2026-05-03T13:02:44Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-10T14:47:54.917408Z"},"links":{"citing_paper":"/paper/2605.01858"},"observation_digest":"sha256:1ed6d5c441700cdbb4cd9a986b743e300c62b4959748331725041a150396330e","observation_id":"ca99b3bd-dc4e-42e5-86f6-5928c1a9bfbf","resolution":{"observed_at":"2026-05-11T11:31:03.617054Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.17453","last_updated":"2024-04-07T00:56:53Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-09-29T17:59:56Z","title":"Efficient Streaming Language Models with Attention Sinks","version":4},"cited_work":{"arxiv_id":"2309.17453","doi":"10.48550/arxiv.2309.17453","metadata_source":"pith","pith_arxiv_id":"2309.17453","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Efficient Streaming Language Models with Attention Sinks","venue":"cs.CL","work_id":"a8d25452-c237-48c9-88a4-682717c3979a","year":2023},"citing_paper":{"arxiv_id":"2605.01858","last_updated":"2026-05-03T13:02:44Z","snapshot_observed_at":"2026-07-06T23:15:01.968194Z","submitted_at":"2026-05-03T13:02:44Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-10T14:47:54.917408Z"},"links":{"cited_paper":"/paper/2309.17453","citing_paper":"/paper/2605.01858"},"observation_digest":"sha256:6e721454897efb12dd5724a35cfdcc23a7df656707ddf318b02926e776d9ee0e","observation_id":"b94c4bb3-fada-4c4a-909f-4b6b4acbe0cd","resolution":{"observed_at":"2026-05-11T11:31:03.621621Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.09608","last_updated":"2025-10-10T17:59:58Z","snapshot_observed_at":"2026-07-06T22:32:22.953630Z","submitted_at":"2025-10-10T17:59:58Z","title":"StreamingVLM: Real-Time Understanding for Infinite Video Streams","version":1},"cited_work":{"arxiv_id":"2510.09608","doi":"10.48550/arxiv.2510.09608","metadata_source":"pith","pith_arxiv_id":"2510.09608","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"StreamingVLM: Real-Time Understanding for Infinite Video Streams","venue":"cs.CV","work_id":"e6785f4f-90d8-45ab-8e34-1a406ea2db88","year":2025},"citing_paper":{"arxiv_id":"2605.01858","last_updated":"2026-05-03T13:02:44Z","snapshot_observed_at":"2026-07-06T23:15:01.968194Z","submitted_at":"2026-05-03T13:02:44Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-10T14:47:54.917408Z"},"links":{"cited_paper":"/paper/2510.09608","citing_paper":"/paper/2605.01858"},"observation_digest":"sha256:cde0a23a05dbcb09d232cb0811c84fa85969dcf192a2685a2eb243ecc6349484","observation_id":"63a5bf8b-564b-42e9-a4b0-665af66bd598","resolution":{"observed_at":"2026-05-17T11:51:33.451476Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.15717","last_updated":"2025-08-21T16:56:29Z","snapshot_observed_at":"2026-07-30T22:19:01.844061Z","submitted_at":"2025-08-21T16:56:29Z","title":"StreamMem: Query-Agnostic KV Cache Memory for Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":"2508.15717","doi":"10.48550/arxiv.2508.15717","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.15717","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"Streammem: Query-agnostic kv cache memory for stream- ing video understanding.arXiv preprint arXiv:2508.15717","venue":null,"work_id":"752bf839-d545-4378-a68d-4a958ba10cee","year":2025},"citing_paper":{"arxiv_id":"2605.01858","last_updated":"2026-05-03T13:02:44Z","snapshot_observed_at":"2026-07-06T23:15:01.968194Z","submitted_at":"2026-05-03T13:02:44Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-10T14:47:54.917408Z"},"links":{"cited_paper":"/paper/2508.15717","citing_paper":"/paper/2605.01858"},"observation_digest":"sha256:80aa28a2a6beb168999135a2a6a54afa96882bd8ecc2ae6a7a24b9098b8c3c02","observation_id":"f4b025e5-d85a-49d5-9bda-3efef210333c","resolution":{"observed_at":"2026-05-11T11:31:03.635026Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-07-06T18:29:30.075334Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2605.01858","last_updated":"2026-05-03T13:02:44Z","snapshot_observed_at":"2026-07-06T23:15:01.968194Z","submitted_at":"2026-05-03T13:02:44Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-10T14:47:54.917408Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2605.01858"},"observation_digest":"sha256:f58b10df1dbe98e8d93365a564777ea1ed465e17e97c3eca7d78c6798bb43902","observation_id":"09c9c0bb-1fff-461f-aa8e-83141dc06e75","resolution":{"observed_at":"2026-05-11T11:31:03.562532Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"5bd94f06-d972-4ac0-a4fa-5fb96d69b9d9","year":2025},"citing_paper":{"arxiv_id":"2605.01858","last_updated":"2026-05-03T13:02:44Z","snapshot_observed_at":"2026-07-06T23:15:01.968194Z","submitted_at":"2026-05-03T13:02:44Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-10T14:47:54.917408Z"},"links":{"citing_paper":"/paper/2605.01858"},"observation_digest":"sha256:e47a26c0b284dbc63d1cef8b6b537fe20a4ac61e088d26e665b2a3d90e924036","observation_id":"df74d09c-8af0-4303-a424-24aa76a10d28","resolution":{"observed_at":"2026-05-18T11:46:19.673538Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2605.01858","last_updated":"2026-05-03T13:02:44Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T23:15:01.968194Z","submitted_at":"2026-05-03T13:02:44Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding"},"reference_resolution":{"displayed":19,"state_counts":{"malformed_identifier":0,"metadata_mismatch":5,"parse_uncertain":0,"unresolved":1,"verified_exact":12,"verified_fuzzy":1},"total_outbound_references":19},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"thesis":"As of 2 August 2026, this Paper Citation Record lists 19 of 19 outbound references and 1 inbound Pith citation observation for arXiv:2605.01858."}