{"as_of":"2026-08-22T12:36:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:87c7baecf4d1da2533f4576414e5938fd21ab9e833463f45eef76a8343ca3f2c","coverage":[{"denominator":46,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":46,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T16:45:23.455170Z","state":"measured"},{"denominator":46,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":46,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-22T06:32:14.747728+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2608.07009/citation-record","integrity":"/paper/2608.07009/integrity","json":"/paper/2608.07009/citation-record.json","paper":"/paper/2608.07009"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:45:23.291458Z","title":"LongBench v2: Towards deeper understanding and reasoning on realistic long-context multitasks","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-10T16:45:23.291458Z"},"links":{"citing_paper":"/paper/2608.07009"},"observation_digest":"sha256:11f80ec49d753895ffc65667659b1d9aae9b98f5f64488dd02d9e50fb83bdb6d","observation_id":"08450613-eb6a-43d6-aae1-f173b5525e8e","resolution":{"observed_at":"2026-08-10T16:45:23.291458Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:45:23.295722Z","title":"IndexCache: Accelerating sparse attention via cross-layer index reuse, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-10T16:45:23.295722Z"},"links":{"citing_paper":"/paper/2608.07009"},"observation_digest":"sha256:4feee27a5b0b85307b12d6613fe41c82efa844a2c9cff447dae2c2d4974d04eb","observation_id":"1b827acd-2743-44e3-9406-16faa4e31499","resolution":{"observed_at":"2026-08-10T16:45:23.295722Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:45:23.299171Z","title":null,"venue":null,"work_id":null,"year":1966},"citing_paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-10T16:45:23.299171Z"},"links":{"citing_paper":"/paper/2608.07009"},"observation_digest":"sha256:2e7e3338cf6fa93fe221ded333a7586bbd6e110867df9d585fd7d1f2531ccbfc","observation_id":"18eecb4a-35bc-4e38-9ce3-169bcea1ace4","resolution":{"observed_at":"2026-08-10T16:45:23.299171Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:45:23.302928Z","title":"Peters, and Arman Cohan","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-10T16:45:23.302928Z"},"links":{"citing_paper":"/paper/2608.07009"},"observation_digest":"sha256:0dbfd0e3dfe837b4539a5c25eb5702a22df9729337adc36b5cf1a2c96e91c65c","observation_id":"63cddaf1-c477-4db0-9da6-49064ca8accf","resolution":{"observed_at":"2026-08-10T16:45:23.302928Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:45:24.175310Z","title":"ArkVale: Efficient generative LLM inference with recallable key-value eviction","venue":null,"work_id":"2c4b2b84-bf4f-4d69-9c13-4dad732005c0","year":2024},"citing_paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-10T16:45:23.310628Z"},"links":{"citing_paper":"/paper/2608.07009"},"observation_digest":"sha256:7255c5ab2eff7a2b860e1043e3b1a3151e8b027866ed47b3b9a9f8b1c0186bf9","observation_id":"f4fe85e3-d7ab-442d-8289-be80dff3b30e","resolution":{"observed_at":"2026-08-10T16:45:24.178841Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2512.10576","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:45:23.803128Z","title":"ESS: An offload- centric latent-cache management architecture for DeepSeek-V3.2-Exp, 2025","venue":null,"work_id":"984c21fb-6c8b-4884-bca1-08533d6d64c8","year":2025},"citing_paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-10T16:45:23.314548Z"},"links":{"citing_paper":"/paper/2608.07009"},"observation_digest":"sha256:8498702c733a8fe505965f5bd9eb1616da90fb6f2d76d44869a8189763427c82","observation_id":"0dd0d590-d780-4eb8-922c-7edb139f8458","resolution":{"observed_at":"2026-08-10T16:45:23.810177Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:45:24.163628Z","title":"MagicPIG: LSH sampling for efficient LLM generation","venue":null,"work_id":"9306f1d9-780f-4e23-ab6e-e0df9abe75b7","year":2025},"citing_paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-10T16:45:23.318523Z"},"links":{"citing_paper":"/paper/2608.07009"},"observation_digest":"sha256:850cf09ee0a9083ca5bf944cd4d9b6fa44a7dc19af138453135d00c8b5ef6669","observation_id":"206a681e-7d02-48f3-a697-3ce2dc7cc6f7","resolution":{"observed_at":"2026-08-10T16:45:24.167750Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:45:24.152154Z","title":"FlashAttention-2: Faster attention with better parallelism and work partitioning","venue":null,"work_id":"be1c8517-0e69-461c-a033-17ceb791fcff","year":2024},"citing_paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-10T16:45:23.321833Z"},"links":{"citing_paper":"/paper/2608.07009"},"observation_digest":"sha256:69160cecb400b87c09ba1b5263bcb4a04f9d6e400d8a1ef4d4f49d3d78014415","observation_id":"bf2644b7-59bd-4cd7-9575-b6aee5ef0486","resolution":{"observed_at":"2026-08-10T16:45:24.156172Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:45:24.140991Z","title":"Fu, Stefano Ermon, Atri Rudra, and Christopher Ré","venue":null,"work_id":"034a76b2-046d-44c9-af0f-80261d3a1fbf","year":2022},"citing_paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-10T16:45:23.325223Z"},"links":{"citing_paper":"/paper/2608.07009"},"observation_digest":"sha256:3b6f28b4b5607596076b99721739543ef96b01dc393ef20b2b4d6ae509cc3e97","observation_id":"443f7eff-8486-4d41-9538-da640552c794","resolution":{"observed_at":"2026-08-10T16:45:24.145252Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:45:24.128641Z","title":"DeepSeek-V3.2: Efficient reasoning & agentic AI","venue":null,"work_id":"87133309-6d1d-46d7-a183-7cfdc9e4a4de","year":2025},"citing_paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-10T16:45:23.328699Z"},"links":{"citing_paper":"/paper/2608.07009"},"observation_digest":"sha256:81b040c297a29456a9c00189168f96b9bf0c74ed46b4fa60540d0b701bd3f2da","observation_id":"950f44f3-a46f-4438-a616-2d5359620b94","resolution":{"observed_at":"2026-08-10T16:45:24.132858Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2512.02556","last_updated":"2025-12-02T09:25:14Z","snapshot_observed_at":"2026-08-17T01:58:07.850738Z","submitted_at":"2025-12-02T09:25:14Z","title":"DeepSeek-V3.2: Pushing the Frontier of Open Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2512.02556","snapshot_observed_at":"2026-08-10T16:45:23.332283Z","title":"DeepSeek-V3.2: Pushing the frontier of open large language models, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-10T16:45:23.332283Z"},"links":{"cited_paper":"/paper/2512.02556","citing_paper":"/paper/2608.07009"},"observation_digest":"sha256:c4d314784184def4a01856891b3265a3ca2db0f5c65e4e9120ef612ffce2c15e","observation_id":"51644309-8908-4f0b-8327-d9b009988433","resolution":{"observed_at":"2026-08-10T16:45:23.332283Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:45:23.335996Z","title":"DeepSeek-V4: Towards highly efficient million-token context intelligence, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-10T16:45:23.335996Z"},"links":{"citing_paper":"/paper/2608.07009"},"observation_digest":"sha256:a92fb100cdda28ea2c8a13f7da5526410c2443e0f69109d01fd8606cbf48d474","observation_id":"87e82dd7-d097-4c32-a51c-b445c0fcbad2","resolution":{"observed_at":"2026-08-10T16:45:23.335996Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:45:24.116563Z","title":"Cost-efficient large language model serving for multi-turn conversations with CachedAttention","venue":null,"work_id":"b568f78f-34e5-4117-8c5a-7f21af3d08e4","year":2024},"citing_paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-10T16:45:23.339727Z"},"links":{"citing_paper":"/paper/2608.07009"},"observation_digest":"sha256:d1cffac943e4a4fbd77a07f0f7dfe46e4a477382f8f0b7024d4e696325f7442f","observation_id":"6639af0c-5fa4-4eca-b8e8-d67202168ffe","resolution":{"observed_at":"2026-08-10T16:45:24.120914Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2602.15763","last_updated":"2026-02-24T10:44:44Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-02-17T17:50:56Z","title":"GLM-5: from Vibe Coding to Agentic Engineering","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2602.15763","snapshot_observed_at":"2026-08-10T16:45:23.343078Z","title":"GLM-5: from vibe coding to agentic engineering, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-10T16:45:23.343078Z"},"links":{"cited_paper":"/paper/2602.15763","citing_paper":"/paper/2608.07009"},"observation_digest":"sha256:a85842c867f07cfb5efaec2a03775d48862a7357886035588d3ddd3768de831f","observation_id":"db37646f-9abf-42aa-b2a3-918176736f6b","resolution":{"observed_at":"2026-08-10T16:45:23.343078Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.11421","last_updated":"2024-03-18T02:30:23Z","snapshot_observed_at":"2026-08-20T04:11:41.116251Z","submitted_at":"2024-03-18T02:30:23Z","title":"FastDecode: High-Throughput GPU-Efficient LLM Serving using Heterogeneous Pipelines","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.11421","snapshot_observed_at":"2026-08-10T16:45:23.346750Z","title":"FastDecode: High-throughput GPU-efficient LLM serving using heterogeneous pipelines, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-10T16:45:23.346750Z"},"links":{"cited_paper":"/paper/2403.11421","citing_paper":"/paper/2608.07009"},"observation_digest":"sha256:a897d86e45a1fdfef6fc876f4ca4654dfee364ee8dc77f578b15223a76717fa1","observation_id":"b101c3db-c662-4e7b-aa80-9a0f26e7f45c","resolution":{"observed_at":"2026-08-10T16:45:23.346750Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:45:24.105626Z","title":"NEO: Saving GPU memory crisis with CPU offloading for online LLM inference","venue":null,"work_id":"49428982-0022-4159-8546-162bba343d92","year":2025},"citing_paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-10T16:45:23.350702Z"},"links":{"citing_paper":"/paper/2608.07009"},"observation_digest":"sha256:2c24cc8b830eebcaacf96f00f4a594d10c53ea36cd12d0b87a82b7ea20aa4eb1","observation_id":"617066a6-f89a-486d-afbc-84989ba4f922","resolution":{"observed_at":"2026-08-10T16:45:24.109280Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:45:23.354354Z","title":"Reformer: The efficient transformer","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-10T16:45:23.354354Z"},"links":{"citing_paper":"/paper/2608.07009"},"observation_digest":"sha256:d3f1c38db1161924af46a2a0f4f118e6ae58c493c16a66be7dd6def2605f8deb","observation_id":"95365699-59c1-4ad6-8de1-4cceda3b1a08","resolution":{"observed_at":"2026-08-10T16:45:23.354354Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:45:23.358099Z","title":"Gonzalez, Hao Zhang, and Ion Stoica","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-10T16:45:23.358099Z"},"links":{"citing_paper":"/paper/2608.07009"},"observation_digest":"sha256:a527b4cbf24ecf104401bf3491f74b6c1d713047b4bc2501a1ddf4662dc47785","observation_id":"7b6669bd-e03c-467f-99d5-3191d56b2bbe","resolution":{"observed_at":"2026-08-10T16:45:23.358099Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:45:24.087346Z","title":"InfiniGen: Efficient generative inference of large language models with dynamic KV cache management","venue":null,"work_id":"d0ba6348-5642-4afb-89a6-eb365d78d326","year":2024},"citing_paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-10T16:45:23.361753Z"},"links":{"citing_paper":"/paper/2608.07009"},"observation_digest":"sha256:7b046e5f8ca9bcf4900f7face8afd27238eb7bae273527627470cdf4f7f224cd","observation_id":"c3219acc-acf7-418a-98fd-1dfa903ad32d","resolution":{"observed_at":"2026-08-10T16:45:24.091321Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:45:24.075476Z","title":"SnapKV: LLM knows what you are looking for before generation","venue":null,"work_id":"5d3f79a0-ee70-4c07-925e-8cdd8a7532b1","year":2024},"citing_paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-10T16:45:23.365570Z"},"links":{"citing_paper":"/paper/2608.07009"},"observation_digest":"sha256:94ca0beea8c4735cdbc7accecee3956ec7f3da8e04733d00299421c2542be856","observation_id":"037b387f-a27f-4a2e-85ad-31ae52571801","resolution":{"observed_at":"2026-08-10T16:45:24.079545Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:45:24.064250Z","title":"ECHO: Efficient KV cache offloading with lossless prefetching for serving native sparse attention LLMs","venue":null,"work_id":"85b27afc-57e3-42ea-bcd2-040a2b31d886","year":2026},"citing_paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-10T16:45:23.368982Z"},"links":{"citing_paper":"/paper/2608.07009"},"observation_digest":"sha256:b3dd5be065aaaab77592d18c8a89b8eba543276d41a6cc22d505e8817fdef056","observation_id":"a899c11b-847a-401d-ab04-7ac5315e7d33","resolution":{"observed_at":"2026-08-10T16:45:24.068201Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:45:24.053796Z","title":"KIVI: A tuning-free asymmetric 2bit quantization for KV cache","venue":null,"work_id":"30578f37-3d5f-40f9-9959-1f49ede603ef","year":2024},"citing_paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-10T16:45:23.372519Z"},"links":{"citing_paper":"/paper/2608.07009"},"observation_digest":"sha256:4b780c8f992487b6b21c989caaf301ae334b7e3b72ca227e261cac9763e82599","observation_id":"18955ac0-7ab2-4126-9599-a8b5f22a4dc1","resolution":{"observed_at":"2026-08-10T16:45:24.057458Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:45:24.042220Z","title":"NVIDIA GH200 Grace Hopper superchip","venue":null,"work_id":"4329391f-1808-4ddb-a3b2-9347ba79982e","year":2026},"citing_paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-10T16:45:23.375879Z"},"links":{"citing_paper":"/paper/2608.07009"},"observation_digest":"sha256:eaf0a02b3ad2fceb5eb909024415a98324c0b5d0979b422d8ff3fc0be84fc115","observation_id":"e5d2f803-6b22-43b3-9283-ec291a443307","resolution":{"observed_at":"2026-08-10T16:45:24.046620Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:45:24.030149Z","title":"Splitwise: Efficient generative LLM inference using phase splitting","venue":null,"work_id":"6d3594c5-fae2-4692-9660-babb718bcd84","year":null},"citing_paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-10T16:45:23.379612Z"},"links":{"citing_paper":"/paper/2608.07009"},"observation_digest":"sha256:4cd3d74103a9d6591906f87a047b697082531388ae2882585e1295f7fad1c568","observation_id":"1b2b2c5b-985b-4edd-acd9-cb1afae4eca1","resolution":{"observed_at":"2026-08-10T16:45:24.034020Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:45:24.018842Z","title":"Mooncake: Trading more storage for less computation—a KVCache-centric architecture for serving LLM chatbot","venue":null,"work_id":"e2ed2bdd-3aa1-4a60-9630-e55e7716c2fe","year":2025},"citing_paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-10T16:45:23.387043Z"},"links":{"citing_paper":"/paper/2608.07009"},"observation_digest":"sha256:1f306dd039c1878b249e7765cdb6ccc6cfb2eea9552327fe735a43909a26c4b0","observation_id":"2c8afc0a-6b0f-4cb4-bfab-05e645e472c4","resolution":{"observed_at":"2026-08-10T16:45:24.023147Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:45:24.007711Z","title":"Qwen3-30B-A3B-Thinking-2507","venue":null,"work_id":"1eacb2a6-0cf7-468f-88b4-c5cb91b85a57","year":2025},"citing_paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-10T16:45:23.390523Z"},"links":{"citing_paper":"/paper/2608.07009"},"observation_digest":"sha256:fac16365ddb564b6f379e15cfe4929eeb326f207fe1c0bd300da95ae3030dd4d","observation_id":"47246dca-dccc-402b-9779-3fcad698a246","resolution":{"observed_at":"2026-08-10T16:45:24.011712Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:45:23.996439Z","title":"Bench serving guide","venue":null,"work_id":"9fc94c94-e549-4980-83e2-b0ea16f450e5","year":2026},"citing_paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-10T16:45:23.393999Z"},"links":{"citing_paper":"/paper/2608.07009"},"observation_digest":"sha256:1f86c4057211d2be89b707a40c3fd6ceae0dbe850ed536877d7c531642942b24","observation_id":"39509634-85f7-41c8-a6c7-7517926305b4","resolution":{"observed_at":"2026-08-10T16:45:23.999816Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:45:23.983058Z","title":"HiSparse: Hierarchical sparse attention","venue":null,"work_id":"771e60ea-d851-4c80-a8ff-b049cef4ec4a","year":null},"citing_paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-10T16:45:23.397212Z"},"links":{"citing_paper":"/paper/2608.07009"},"observation_digest":"sha256:9a4991a4ed82b7a71a1b2000e089a13bdffa166f447b533197a3a656ab59b024","observation_id":"4b0eecd4-41e1-4e3b-8ef8-a7534b2d602c","resolution":{"observed_at":"2026-08-10T16:45:23.987113Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:45:23.960529Z","title":"FlexGen: High-throughput generative inference of large language models with a single GPU","venue":null,"work_id":"26ee3748-5648-4032-97b8-3db389942640","year":2023},"citing_paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-10T16:45:23.403797Z"},"links":{"citing_paper":"/paper/2608.07009"},"observation_digest":"sha256:8db50a3e9cfc9012c58ed81367d927de08c6dac43b6f2717a777ca1cd2591668","observation_id":"d841648c-33c0-4963-9330-7a92637bf412","resolution":{"observed_at":"2026-08-10T16:45:23.964365Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:45:23.948767Z","title":"ShadowKV: KV cache in shadows for high-throughput long-context LLM inference","venue":null,"work_id":"9929ceaf-b1bd-4347-a15a-88f9044c81d4","year":2025},"citing_paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-10T16:45:23.406916Z"},"links":{"citing_paper":"/paper/2608.07009"},"observation_digest":"sha256:821de20a5064d728817739cd97e51132297982053e4bcfc5d247d2dcda981fde","observation_id":"cbefa9c6-2a90-43ae-bb7a-7eedeebdf0c6","resolution":{"observed_at":"2026-08-10T16:45:23.952778Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:45:23.937302Z","title":"Quest: Query-aware sparsity for efficient long-context LLM inference","venue":null,"work_id":"896e1a3f-aeb7-4682-82b4-585e508ad3a6","year":2024},"citing_paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-10T16:45:23.410330Z"},"links":{"citing_paper":"/paper/2608.07009"},"observation_digest":"sha256:84fda018dd76b21e032ad2916b9cc59321a0fb262c990b9f8ad6c45d800fd7bd","observation_id":"2ff478fc-ffba-4d28-be6f-4d8d73867a95","resolution":{"observed_at":"2026-08-10T16:45:23.940899Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04768","last_updated":"2020-06-14T08:15:54Z","snapshot_observed_at":"2026-07-06T09:27:03.809621Z","submitted_at":"2020-06-08T17:37:52Z","title":"Linformer: Self-Attention with Linear Complexity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04768","snapshot_observed_at":"2026-08-10T16:45:23.414556Z","title":"Li, Madian Khabsa, Han Fang, and Hao Ma","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-10T16:45:23.414556Z"},"links":{"cited_paper":"/paper/2006.04768","citing_paper":"/paper/2608.07009"},"observation_digest":"sha256:35837f318d17ae59640f2d0b3d2759189f7caae4af5a40748b3b93a0c84d51a9","observation_id":"610d0934-c4f7-4171-906e-006158d27888","resolution":{"observed_at":"2026-08-10T16:45:23.414556Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:45:23.419073Z","title":"Efficient streaming language models with attention sinks","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-10T16:45:23.419073Z"},"links":{"citing_paper":"/paper/2608.07009"},"observation_digest":"sha256:9493437140a05163d5cf9024c749130aca4585facaae22e19c9df17ec8527c34","observation_id":"b2f6430f-d3a9-4593-8931-bccccdfc123d","resolution":{"observed_at":"2026-08-10T16:45:23.419073Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:45:23.918905Z","title":"SGLang HiCache: Fast hierarchical KV caching with your fa- vorite storage backends","venue":null,"work_id":"a5b3fd65-da38-419a-9a8c-5a39688945ba","year":2025},"citing_paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-10T16:45:23.423045Z"},"links":{"citing_paper":"/paper/2608.07009"},"observation_digest":"sha256:a4c299dd603d8a0e1477a70e793e2bde8a4f60a83ea0e9e06d70e847b9a81a0c","observation_id":"73ef9abf-fc94-4a59-9ae7-35d9ac0249d9","resolution":{"observed_at":"2026-08-10T16:45:23.924134Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:45:23.906622Z","title":"HiSparse: Turbocharging sparse attention with hierarchical memory","venue":null,"work_id":"e3a1565e-c6e8-46cd-82d8-55072757866b","year":2026},"citing_paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-10T16:45:23.426667Z"},"links":{"citing_paper":"/paper/2608.07009"},"observation_digest":"sha256:a557f9c64f7c7ee52bbf57b8ed471e33699858692620f26bf519c6e6ff29b9b0","observation_id":"229d2781-e7f0-422d-9803-08c083ce68a3","resolution":{"observed_at":"2026-08-10T16:45:23.910968Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.18572","last_updated":"2025-08-26T00:09:03Z","snapshot_observed_at":"2026-08-18T12:35:51.096042Z","submitted_at":"2025-08-26T00:09:03Z","title":"Strata: Hierarchical Context Caching for Long Context Language Model Serving","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.18572","snapshot_observed_at":"2026-08-10T16:45:23.429982Z","title":"Strata: Hierarchical context caching for long context language model serving","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-10T16:45:23.429982Z"},"links":{"cited_paper":"/paper/2508.18572","citing_paper":"/paper/2608.07009"},"observation_digest":"sha256:4a1439e24e9e8a71e157c7d2d09f51bb3b3da0faccccf52cbd915bac7f282658","observation_id":"cf222180-b25d-4802-9352-9aabc53fcfa9","resolution":{"observed_at":"2026-08-10T16:45:23.429982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.09388","snapshot_observed_at":"2026-08-10T16:45:23.433719Z","title":"Qwen3 technical report, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-10T16:45:23.433719Z"},"links":{"cited_paper":"/paper/2505.09388","citing_paper":"/paper/2608.07009"},"observation_digest":"sha256:08f9296313abf8df6c0be1e2476767392dc4270b05a94c3e081036cf01631474","observation_id":"d4644d1c-4603-434f-9620-4e0d0f09fdcb","resolution":{"observed_at":"2026-08-10T16:45:23.433719Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:45:23.437624Z","title":"Native sparse attention: Hardware-aligned and natively trainable sparse attention","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-10T16:45:23.437624Z"},"links":{"citing_paper":"/paper/2608.07009"},"observation_digest":"sha256:3c8ff6aa529ad56167cd84e69fea1014412994af3393321aedb2483965c1dbc0","observation_id":"78ddc875-30a7-4cce-966b-3257618e3929","resolution":{"observed_at":"2026-08-10T16:45:23.437624Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:45:23.896165Z","title":"GLM-5.2: Built for long-horizon tasks","venue":null,"work_id":"05a8ff2e-b74c-44cf-b5ae-9888be46e6cd","year":2026},"citing_paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-10T16:45:23.441276Z"},"links":{"citing_paper":"/paper/2608.07009"},"observation_digest":"sha256:a38310f09ec5030b4870cd2de5c1cde3ce0d76f8c4f6355679d43cd6560e1a21","observation_id":"2c6683b0-4b82-4e1d-8838-331b0e82d336","resolution":{"observed_at":"2026-08-10T16:45:23.899561Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:45:23.444854Z","title":"PQCache: Product quantization-based KVCache for long context LLM inference.Proceedings of the ACM on Management of Data, 3(3):201:1–201:30, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-10T16:45:23.444854Z"},"links":{"citing_paper":"/paper/2608.07009"},"observation_digest":"sha256:4e8ff41790d1a5a015500c615177d716dd05dd28c6ac56dbf9f5ab984c220243","observation_id":"126ae745-301c-44e3-8cb1-1c20ab255f32","resolution":{"observed_at":"2026-08-10T16:45:23.444854Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:45:23.448711Z","title":"H2O: Heavy-hitter oracle for efficient generative inference of large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-10T16:45:23.448711Z"},"links":{"citing_paper":"/paper/2608.07009"},"observation_digest":"sha256:c9d46e8df0c45644f795daae3f6333f7a31568cd55e057cda94dbce9d38e2194","observation_id":"dbdb14c9-42e4-49cf-beff-ad7264c493c0","resolution":{"observed_at":"2026-08-10T16:45:23.448711Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:45:23.878830Z","title":"Gonzalez, Clark W","venue":null,"work_id":"77b92e25-6aab-4bc9-9407-73ca323e1379","year":2024},"citing_paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-10T16:45:23.451960Z"},"links":{"citing_paper":"/paper/2608.07009"},"observation_digest":"sha256:988da29a4fe96c5594566636ca3e1b647acb538211a42f81e9190ae85295b767","observation_id":"1b4e21e5-f645-416c-a7cb-0cdc26cc8fa7","resolution":{"observed_at":"2026-08-10T16:45:23.882352Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:45:23.866927Z","title":"DistServe: Disaggregating prefill and decoding for goodput-optimized large language model serving","venue":null,"work_id":"dfab5ed0-6e5a-47ec-a48d-09ea05667572","year":2024},"citing_paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-10T16:45:23.455170Z"},"links":{"citing_paper":"/paper/2608.07009"},"observation_digest":"sha256:c2993e5470d16c56cb343be4fe824270dbf9f796fc8bef67963cc31228dc4060","observation_id":"d6f3142d-bdb0-409f-9231-eb31c460ab82","resolution":{"observed_at":"2026-08-10T16:45:23.870987Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2004.05150","last_updated":"2020-12-02T17:52:35Z","snapshot_observed_at":"2026-07-31T17:17:17.205582Z","submitted_at":"2020-04-10T17:54:09Z","title":"Longformer: The Long-Document Transformer","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2004.05150","snapshot_observed_at":"2026-08-10T16:45:23.306275Z","title":null,"venue":null,"work_id":null,"year":2004},"citing_paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management","version":1},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-10T16:45:23.306275Z"},"links":{"cited_paper":"/paper/2004.05150","citing_paper":"/paper/2608.07009"},"observation_digest":"sha256:c7ca7505f7c953a81751630d5ad23ed3e04b956cdb83a3afe92b4e5b6741ccc2","observation_id":"7ea78599-617f-4bfd-82cd-0322b5c4281c","resolution":{"observed_at":"2026-08-10T16:45:23.306275Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:45:23.383477Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-10T16:45:23.383477Z"},"links":{"citing_paper":"/paper/2608.07009"},"observation_digest":"sha256:ad6a6d236d70417f5a48b7d75dbba7833fe2fa7bfcc5e479ebddb2fba780b379","observation_id":"ccbcf82a-b759-4eb3-8e41-62a5b8a1444d","resolution":{"observed_at":"2026-08-10T16:45:23.383477Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:45:23.971717Z","title":"Accessed 2026-05-04","venue":null,"work_id":"8d0533b5-a6b5-4b12-a9e4-5ad99feca76e","year":2026},"citing_paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management","version":1},"reference_index":2026,"source":"pdf_text","source_observed_at":"2026-08-10T16:45:23.400534Z"},"links":{"citing_paper":"/paper/2608.07009"},"observation_digest":"sha256:6a246bb0c59009a479a1e0c9d540661890ca339c63e3c8c696b810d44f27d234","observation_id":"d260293f-3b99-47e9-909d-a75a55b1efae","resolution":{"observed_at":"2026-08-10T16:45:23.975469Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2608.07009","last_updated":"2026-08-07T09:22:17Z","latest_version":1,"primary_category":"cs.DC","snapshot_observed_at":"2026-08-18T08:36:39.548295Z","submitted_at":"2026-08-07T09:22:17Z","title":"HiSparse: Scaling Sparse-Attention Decoding with Hierarchical KV Cache Management"},"reference_resolution":{"displayed":46,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":19,"verified_exact":1,"verified_fuzzy":26},"total_outbound_references":46},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"thesis":"As of 22 August 2026, this Paper Citation Record lists 46 of 46 outbound references and 0 inbound Pith citation observations for arXiv:2608.07009."}