{"as_of":"2026-08-08T11:40:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:e5923b39e3fe84b20eb88848a2fdb2ab7ba0c99ac5ef4db48c0c978e5f462638","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":10,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":10,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":10,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":10,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:48:00.017529Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-01T08:15:32.438184Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2402.15220","last_updated":"2024-08-01T07:51:25Z","snapshot_observed_at":"2026-08-04T06:24:10.295170Z","submitted_at":"2024-02-23T09:29:19Z","title":"ChunkAttention: Efficient Self-Attention with Prefix-Aware KV Cache and Two-Phase Partition","version":4},"cited_work":{"arxiv_id":"2402.15220","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.15220","snapshot_observed_at":"2026-07-01T08:15:32.438184Z","title":"Chunkattention: Efficient self-attention with prefix-aware kv cache and two-phase partition","venue":null,"work_id":"f555b084-a556-4195-9106-ada09956a485","year":2024},"citing_paper":{"arxiv_id":"2312.07104","last_updated":"2024-06-06T00:10:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-12T09:34:27Z","title":"SGLang: Efficient Execution of Structured Language Model Programs","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-05-12T08:20:01.011625Z"},"links":{"cited_paper":"/paper/2402.15220","citing_paper":"/paper/2312.07104"},"observation_digest":"sha256:e99fd48a9005232455585c763729cb928bb62068c2e068ef5c832afde8a9feb9","observation_id":"15a4b487-9e26-478c-9c62-a13b250ca0c4","resolution":{"observed_at":"2026-05-12T08:20:01.172272Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.15220","last_updated":"2024-08-01T07:51:25Z","snapshot_observed_at":"2026-08-04T06:24:10.295170Z","submitted_at":"2024-02-23T09:29:19Z","title":"ChunkAttention: Efficient Self-Attention with Prefix-Aware KV Cache and Two-Phase Partition","version":4},"cited_work":{"arxiv_id":"2402.15220","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.15220","snapshot_observed_at":"2026-07-01T08:15:32.438184Z","title":"Chunkattention: Efficient self-attention with prefix-aware kv cache and two-phase partition","venue":null,"work_id":"f555b084-a556-4195-9106-ada09956a485","year":2024},"citing_paper":{"arxiv_id":"2504.15965","last_updated":"2025-04-23T13:47:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-22T15:05:04Z","title":"From Human Memory to AI Memory: A Survey on Memory Mechanisms in the Era of LLMs","version":2},"reference_index":130,"source":"pdf_text","source_observed_at":"2026-05-17T11:05:09.588491Z"},"links":{"cited_paper":"/paper/2402.15220","citing_paper":"/paper/2504.15965"},"observation_digest":"sha256:5b7c4ead423e5f90c57ee224d9f427953fafd0a93cdf9c6097411cd9ebc9cdb0","observation_id":"414770d3-96ba-4384-b9bf-fff96fb5381f","resolution":{"observed_at":"2026-05-17T11:05:09.861240Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.15220","last_updated":"2024-08-01T07:51:25Z","snapshot_observed_at":"2026-08-04T06:24:10.295170Z","submitted_at":"2024-02-23T09:29:19Z","title":"ChunkAttention: Efficient Self-Attention with Prefix-Aware KV Cache and Two-Phase Partition","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.15220","snapshot_observed_at":"2026-08-07T14:48:00.017529Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17694","last_updated":"2026-03-28T10:14:51Z","snapshot_observed_at":"2026-08-07T14:40:02.904487Z","submitted_at":"2025-05-23T10:03:28Z","title":"CoDec: Prefix-Shared Decoding Kernel for LLMs","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-07T14:48:00.017529Z"},"links":{"cited_paper":"/paper/2402.15220","citing_paper":"/paper/2505.17694"},"observation_digest":"sha256:72ad2e8aa77639aa186b6ae48d3145ae479bf1ef207b1a42da7054af963d16ee","observation_id":"1b793694-4bad-4346-8fd6-2429227ed1ec","resolution":{"observed_at":"2026-08-07T14:48:00.017529Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.15220","last_updated":"2024-08-01T07:51:25Z","snapshot_observed_at":"2026-08-04T06:24:10.295170Z","submitted_at":"2024-02-23T09:29:19Z","title":"ChunkAttention: Efficient Self-Attention with Prefix-Aware KV Cache and Two-Phase Partition","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.15220","snapshot_observed_at":"2026-08-07T06:03:19.362463Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.11104","last_updated":"2025-06-06T20:24:36Z","snapshot_observed_at":"2026-08-07T20:41:32.031099Z","submitted_at":"2025-06-06T20:24:36Z","title":"DAM: Dynamic Attention Mask for Long-Context Large Language Model Inference Acceleration","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-07T06:03:19.362463Z"},"links":{"cited_paper":"/paper/2402.15220","citing_paper":"/paper/2506.11104"},"observation_digest":"sha256:d973358f6d332c2c7999a65114b6cf7718de69da234a1d7c9739c97b211428f5","observation_id":"d96aaaba-19ec-46e6-b609-794c52746810","resolution":{"observed_at":"2026-08-07T06:03:19.362463Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.15220","last_updated":"2024-08-01T07:51:25Z","snapshot_observed_at":"2026-08-04T06:24:10.295170Z","submitted_at":"2024-02-23T09:29:19Z","title":"ChunkAttention: Efficient Self-Attention with Prefix-Aware KV Cache and Two-Phase Partition","version":4},"cited_work":{"arxiv_id":"2402.15220","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.15220","snapshot_observed_at":"2026-07-01T08:15:32.438184Z","title":"Chunkattention: Efficient self-attention with prefix-aware kv cache and two-phase partition","venue":null,"work_id":"f555b084-a556-4195-9106-ada09956a485","year":2024},"citing_paper":{"arxiv_id":"2602.09725","last_updated":"2026-05-12T14:12:11Z","snapshot_observed_at":"2026-07-06T22:45:15.978102Z","submitted_at":"2026-02-10T12:29:02Z","title":"Efficient Remote KV Cache Reuse with GPU-native Video Codec","version":3},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-05-16T05:21:04.555356Z"},"links":{"cited_paper":"/paper/2402.15220","citing_paper":"/paper/2602.09725"},"observation_digest":"sha256:b5b3a0bfe6a60eee1fa609aa6291296b965e468de1fced4f3d04b9fdf5dd83b4","observation_id":"bc12ec78-6281-4597-b4d6-568c6019802a","resolution":{"observed_at":"2026-05-16T05:22:22.841918Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.15220","last_updated":"2024-08-01T07:51:25Z","snapshot_observed_at":"2026-08-04T06:24:10.295170Z","submitted_at":"2024-02-23T09:29:19Z","title":"ChunkAttention: Efficient Self-Attention with Prefix-Aware KV Cache and Two-Phase Partition","version":4},"cited_work":{"arxiv_id":"2402.15220","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.15220","snapshot_observed_at":"2026-07-01T08:15:32.438184Z","title":"Chunkattention: Efficient self-attention with prefix-aware kv cache and two-phase partition","venue":null,"work_id":"f555b084-a556-4195-9106-ada09956a485","year":2024},"citing_paper":{"arxiv_id":"2605.03375","last_updated":"2026-05-05T05:33:11Z","snapshot_observed_at":"2026-07-06T23:16:20.322265Z","submitted_at":"2026-05-05T05:33:11Z","title":"Tutti: Making SSD-Backed KV Cache Practical for Long-Context LLM Serving","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-09T16:19:33.613685Z"},"links":{"cited_paper":"/paper/2402.15220","citing_paper":"/paper/2605.03375"},"observation_digest":"sha256:a28967cd30e57dca455e962cc14d4faa497813c6ec6e42534a5bc905cc7b081a","observation_id":"48dfb67e-6912-4f0e-86c9-0b19b5f85c36","resolution":{"observed_at":"2026-05-11T16:31:10.450487Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.15220","last_updated":"2024-08-01T07:51:25Z","snapshot_observed_at":"2026-08-04T06:24:10.295170Z","submitted_at":"2024-02-23T09:29:19Z","title":"ChunkAttention: Efficient Self-Attention with Prefix-Aware KV Cache and Two-Phase Partition","version":4},"cited_work":{"arxiv_id":"2402.15220","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.15220","snapshot_observed_at":"2026-07-01T08:15:32.438184Z","title":"Chunkattention: Efficient self-attention with prefix-aware kv cache and two-phase partition","venue":null,"work_id":"f555b084-a556-4195-9106-ada09956a485","year":2024},"citing_paper":{"arxiv_id":"2605.05219","last_updated":"2026-04-17T09:24:58Z","snapshot_observed_at":"2026-07-06T23:17:52.987860Z","submitted_at":"2026-04-17T09:24:58Z","title":"Sparse Prefix Caching for Hybrid and Recurrent LLM Serving","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-10T08:49:24.880528Z"},"links":{"cited_paper":"/paper/2402.15220","citing_paper":"/paper/2605.05219"},"observation_digest":"sha256:2d9ac41b8aae2ebc6a91003336effb819762e1cf64e2b3f79471ee2ddc6ba8bd","observation_id":"29b7daa5-b088-4d81-9f14-c6ead04677a7","resolution":{"observed_at":"2026-05-10T08:53:04.421711Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.15220","last_updated":"2024-08-01T07:51:25Z","snapshot_observed_at":"2026-08-04T06:24:10.295170Z","submitted_at":"2024-02-23T09:29:19Z","title":"ChunkAttention: Efficient Self-Attention with Prefix-Aware KV Cache and Two-Phase Partition","version":4},"cited_work":{"arxiv_id":"2402.15220","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.15220","snapshot_observed_at":"2026-07-01T08:15:32.438184Z","title":"Chunkattention: Efficient self-attention with prefix-aware kv cache and two-phase partition","venue":null,"work_id":"f555b084-a556-4195-9106-ada09956a485","year":2024},"citing_paper":{"arxiv_id":"2605.09735","last_updated":"2026-06-30T03:39:11Z","snapshot_observed_at":"2026-07-06T23:21:49.499951Z","submitted_at":"2026-05-10T20:10:26Z","title":"KV-RM: Regularizing KV-Cache Movement for Static-Graph LLM Serving","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-12T03:57:30.104609Z"},"links":{"cited_paper":"/paper/2402.15220","citing_paper":"/paper/2605.09735"},"observation_digest":"sha256:70f93783a129a461b7812bd2843cc7e0aa785d9e227363eaac4cc96d4c636cf9","observation_id":"65e96e4d-ecde-4f5a-a3cc-0c4a849ee976","resolution":{"observed_at":"2026-05-12T06:51:26.907156Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.15220","last_updated":"2024-08-01T07:51:25Z","snapshot_observed_at":"2026-08-04T06:24:10.295170Z","submitted_at":"2024-02-23T09:29:19Z","title":"ChunkAttention: Efficient Self-Attention with Prefix-Aware KV Cache and Two-Phase Partition","version":4},"cited_work":{"arxiv_id":"2402.15220","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.15220","snapshot_observed_at":"2026-07-01T08:15:32.438184Z","title":"Chunkattention: Efficient self-attention with prefix-aware kv cache and two-phase partition","venue":null,"work_id":"f555b084-a556-4195-9106-ada09956a485","year":2024},"citing_paper":{"arxiv_id":"2605.09735","last_updated":"2026-06-30T03:39:11Z","snapshot_observed_at":"2026-07-06T23:21:49.499951Z","submitted_at":"2026-05-10T20:10:26Z","title":"KV-RM: Regularizing KV-Cache Movement for Static-Graph LLM Serving","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-07-01T08:05:44.256565Z"},"links":{"cited_paper":"/paper/2402.15220","citing_paper":"/paper/2605.09735"},"observation_digest":"sha256:6e5786bdaa41da636ace9ee029ea89387209399e1365f1b824961072eb5bf0a0","observation_id":"f12dfc56-0f28-4b2f-ace6-6a43b00e2c44","resolution":{"observed_at":"2026-07-01T08:15:32.441185Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.15220","last_updated":"2024-08-01T07:51:25Z","snapshot_observed_at":"2026-08-04T06:24:10.295170Z","submitted_at":"2024-02-23T09:29:19Z","title":"ChunkAttention: Efficient Self-Attention with Prefix-Aware KV Cache and Two-Phase Partition","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.15220","snapshot_observed_at":"2026-07-14T10:41:08.967261Z","title":"ChunkAttention: Efficient self-attention with prefix-aware KV cache and two-phase partition,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.10582","last_updated":"2026-07-12T05:35:26Z","snapshot_observed_at":"2026-07-16T23:18:51.454376Z","submitted_at":"2026-07-12T05:35:26Z","title":"MemDecay: Region-Aware KV Cache Eviction for Efficient LLM Agent Inference","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-07-14T10:41:08.967261Z"},"links":{"cited_paper":"/paper/2402.15220","citing_paper":"/paper/2607.10582"},"observation_digest":"sha256:4b94bf8d547a7a55b86e816f3ca2e0cc76eccb4dc393a64319a05fee0f014dad","observation_id":"08a7961a-011b-4b6f-b790-034d015d43d7","resolution":{"observed_at":"2026-07-14T10:41:08.967261Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2402.15220/citation-record","integrity":"/paper/2402.15220/integrity","json":"/paper/2402.15220/citation-record.json","paper":"/paper/2402.15220"},"outbound":[],"paper":{"arxiv_id":"2402.15220","last_updated":"2024-08-01T07:51:25Z","latest_version":4,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-04T06:24:10.295170Z","submitted_at":"2024-02-23T09:29:19Z","title":"ChunkAttention: Efficient Self-Attention with Prefix-Aware KV Cache and Two-Phase Partition"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 10 inbound Pith citation observations for arXiv:2402.15220."}