{"as_of":"2026-08-07T00:57:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:c09a7b6ddeedfe88c31adf42f6444216379984d73ddcf3ce1ae2f489df3360cb","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":45,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":45,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":45,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":45,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T18:11:15.969489Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"pith","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":13,"observed_at":"2026-08-05T02:28:24.338817Z","source":"pith"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":"2407.00079","doi":"10.48550/arxiv.2407.00079","metadata_source":"pith","pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mooncake: A kvcache-centric disaggregated architecture for llm serving","venue":"cs.DC","work_id":"eee8f937-db6b-4c48-887b-960ca80c0d81","year":2024},"citing_paper":{"arxiv_id":"2409.10516","last_updated":"2024-12-31T07:11:00Z","snapshot_observed_at":"2026-07-06T19:16:14.638445Z","submitted_at":"2024-09-16T17:59:52Z","title":"RetrievalAttention: Accelerating Long-Context LLM Inference via Vector Retrieval","version":3},"reference_index":105,"source":"arxiv_source","source_observed_at":"2026-05-18T08:12:01.798459Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2409.10516"},"observation_digest":"sha256:97560583725e0c04ad0fa8e0c391e1336357351fca79a3cffec5f220d05e8a76","observation_id":"352b9913-3e7c-4066-8ad5-660f610f4814","resolution":{"observed_at":"2026-05-18T08:12:02.005662Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":"2407.00079","doi":"10.48550/arxiv.2407.00079","metadata_source":"pith","pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mooncake: A kvcache-centric disaggregated architecture for llm serving","venue":"cs.DC","work_id":"eee8f937-db6b-4c48-887b-960ca80c0d81","year":2024},"citing_paper":{"arxiv_id":"2412.03594","last_updated":"2026-04-22T15:33:51Z","snapshot_observed_at":"2026-07-06T20:01:45.826971Z","submitted_at":"2024-11-29T05:57:37Z","title":"BatchLLM: Optimizing Large Batched LLM Inference with Global Prefix Sharing and Throughput-oriented Token Batching","version":3},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-23T16:57:46.645061Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2412.03594"},"observation_digest":"sha256:0bcfaa61313a259b4222838c9b50ebde8b675ddb6dc849e4651d3c1f5df37300","observation_id":"7528d19c-e4e7-4b6a-b19b-c5c1cff684c4","resolution":{"observed_at":"2026-05-23T16:58:11.891568Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":"2407.00079","doi":"10.48550/arxiv.2407.00079","metadata_source":"pith","pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mooncake: A kvcache-centric disaggregated architecture for llm serving","venue":"cs.DC","work_id":"eee8f937-db6b-4c48-887b-960ca80c0d81","year":2024},"citing_paper":{"arxiv_id":"2502.10248","last_updated":"2025-02-24T10:12:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-14T15:58:10Z","title":"Step-Video-T2V Technical Report: The Practice, Challenges, and Future of Video Foundation Model","version":3},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-05-19T08:02:23.002090Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2502.10248"},"observation_digest":"sha256:6e61cb82bb2dce1de291edce831ad791b856ea6fb2e8f91e50525dd347a8fe1b","observation_id":"1597fe2b-4566-4862-95e8-7fe26d69ce6c","resolution":{"observed_at":"2026-05-19T08:02:23.720450Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":"2407.00079","doi":"10.48550/arxiv.2407.00079","metadata_source":"pith","pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mooncake: A kvcache-centric disaggregated architecture for llm serving","venue":"cs.DC","work_id":"eee8f937-db6b-4c48-887b-960ca80c0d81","year":2024},"citing_paper":{"arxiv_id":"2504.15965","last_updated":"2025-04-23T13:47:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-22T15:05:04Z","title":"From Human Memory to AI Memory: A Survey on Memory Mechanisms in the Era of LLMs","version":2},"reference_index":121,"source":"pdf_text","source_observed_at":"2026-05-17T11:05:09.588491Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2504.15965"},"observation_digest":"sha256:947deb17a5bdc34ca54f9893aedd20f73579751ac9d041002668558d3ddbcba6","observation_id":"b0771189-3fea-4e80-91d1-60c3bcb1ea70","resolution":{"observed_at":"2026-05-17T11:05:09.843818Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":"2407.00079","doi":"10.48550/arxiv.2407.00079","metadata_source":"pith","pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mooncake: A kvcache-centric disaggregated architecture for llm serving","venue":"cs.DC","work_id":"eee8f937-db6b-4c48-887b-960ca80c0d81","year":2024},"citing_paper":{"arxiv_id":"2505.23970","last_updated":"2026-04-11T06:24:23Z","snapshot_observed_at":"2026-08-04T02:14:18.533002Z","submitted_at":"2025-05-29T19:52:44Z","title":"Cache Your Prompt When It's Green: Carbon-Aware Caching for Large Language Model Serving","version":3},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-19T13:14:26.628447Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2505.23970"},"observation_digest":"sha256:817ba478a69f41dce3772db05315b9e9e535dee862515549c0fa89b0bd2d13c8","observation_id":"3a129b94-c23c-47d6-b7e9-d5193030acaa","resolution":{"observed_at":"2026-05-19T13:17:18.476761Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-06T18:11:15.969489Z","title":"Noam Shazeer","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.09019","last_updated":"2025-07-11T20:58:21Z","snapshot_observed_at":"2026-08-06T18:03:49.435059Z","submitted_at":"2025-07-11T20:58:21Z","title":"On Evaluating Performance of LLM Inference Serving Systems","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T18:11:15.969489Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2507.09019"},"observation_digest":"sha256:fb186888ba58ca8f27b825dc0d0e730b2e0810e5dc02ea73b04f037cf7c942bb","observation_id":"9ebb2090-66ea-44ff-9cb0-72c5a1ba1f84","resolution":{"observed_at":"2026-08-06T18:11:15.969489Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-06T17:46:20.294435Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.10150","last_updated":"2025-07-14T10:53:47Z","snapshot_observed_at":"2026-08-06T17:36:00.049852Z","submitted_at":"2025-07-14T10:53:47Z","title":"Past-Future Scheduler for LLM Serving under SLA Guarantees","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T17:46:20.294435Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2507.10150"},"observation_digest":"sha256:bcb38ca8b5cbb599becf40eeca54b21e4af2ef2fb20255eaa77202cca11221d1","observation_id":"2fe5dd24-aa07-44fe-8361-135161d09c58","resolution":{"observed_at":"2026-08-06T17:46:20.294435Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-06T17:06:33.971861Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.11953","last_updated":"2025-07-16T06:39:11Z","snapshot_observed_at":"2026-08-06T16:55:17.391172Z","submitted_at":"2025-07-16T06:39:11Z","title":"IAM: Efficient Inference through Attention Mapping between Different-scale LLMs","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-06T17:06:33.971861Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2507.11953"},"observation_digest":"sha256:ba36b6caac85c920251a1e13692be6b17a21ec4aec77827736c0d190ca53f79e","observation_id":"42bc596e-6c1b-4d5c-90ba-f02346f3abcd","resolution":{"observed_at":"2026-08-06T17:06:33.971861Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":"2407.00079","doi":"10.48550/arxiv.2407.00079","metadata_source":"pith","pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mooncake: A kvcache-centric disaggregated architecture for llm serving","venue":"cs.DC","work_id":"eee8f937-db6b-4c48-887b-960ca80c0d81","year":2024},"citing_paper":{"arxiv_id":"2507.18454","last_updated":"2026-04-15T08:07:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-19T06:37:29Z","title":"Sandwich: Joint Configuration Search and Hot-Switching for Efficient CPU LLM Serving","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-22T15:08:15.328986Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2507.18454"},"observation_digest":"sha256:970c86f6c65c58353d43c5adcfad642b527fab06e7d6b5297c87ff3feab7c803","observation_id":"7e423ebc-4e6b-4f56-876d-b671ecf41ed6","resolution":{"observed_at":"2026-05-22T15:11:43.902548Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":"2407.00079","doi":"10.48550/arxiv.2407.00079","metadata_source":"pith","pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mooncake: A kvcache-centric disaggregated architecture for llm serving","venue":"cs.DC","work_id":"eee8f937-db6b-4c48-887b-960ca80c0d81","year":2024},"citing_paper":{"arxiv_id":"2508.15919","last_updated":"2026-04-23T22:48:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-08-21T18:40:20Z","title":"HFX: Joint Design of Algorithms and Systems for Multi-SLO Serving and Fast Scaling","version":3},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-18T21:39:02.560962Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2508.15919"},"observation_digest":"sha256:63e688733e94fcf603da5c5f4ed375ea223e5d6c285229a40dcf35daebf77c91","observation_id":"0e7aad83-de96-4a2d-a74d-e8aea9d126dd","resolution":{"observed_at":"2026-05-18T21:41:51.752300Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-05T15:45:38.950601Z","title":"Mooncake: A kvcache-centric disaggre- gated architecture for llm serving.arXiv preprint arXiv:2407.00079, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.19559","last_updated":"2025-08-27T04:22:02Z","snapshot_observed_at":"2026-08-06T12:06:42.525601Z","submitted_at":"2025-08-27T04:22:02Z","title":"Taming the Chaos: Coordinated Autoscaling for Heterogeneous and Disaggregated LLM Inference","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-05T15:45:38.950601Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2508.19559"},"observation_digest":"sha256:126fa0b69550b4b0879aa0f9d955e976b204895ae335ad26dba7d8d830623d44","observation_id":"92be0001-d2d4-4b27-b8bb-b84232124710","resolution":{"observed_at":"2026-08-05T15:45:38.950601Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":"2407.00079","doi":"10.48550/arxiv.2407.00079","metadata_source":"pith","pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mooncake: A kvcache-centric disaggregated architecture for llm serving","venue":"cs.DC","work_id":"eee8f937-db6b-4c48-887b-960ca80c0d81","year":2024},"citing_paper":{"arxiv_id":"2510.18586","last_updated":"2026-05-20T07:55:01Z","snapshot_observed_at":"2026-07-06T22:33:43.999976Z","submitted_at":"2025-10-21T12:39:32Z","title":"TokenCake: A KV-Cache-centric Serving Framework for LLM-based Multi-Agent Applications","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-21T21:17:58.275784Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2510.18586"},"observation_digest":"sha256:4d25b734e4af59034a1477e12a86ce1b8e0ab12f64c71fca3a8b8ba4ad328d71","observation_id":"259df76e-5863-4d72-8fd7-f6bd7ca33955","resolution":{"observed_at":"2026-05-21T21:20:38.741005Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":"2407.00079","doi":"10.48550/arxiv.2407.00079","metadata_source":"pith","pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mooncake: A kvcache-centric disaggregated architecture for llm serving","venue":"cs.DC","work_id":"eee8f937-db6b-4c48-887b-960ca80c0d81","year":2024},"citing_paper":{"arxiv_id":"2511.00413","last_updated":"2026-04-23T16:13:03Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-11-01T05:56:49Z","title":"Tree Training: Accelerating Agentic LLMs Training via Shared Prefix Reuse","version":5},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-18T02:04:46.559881Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2511.00413"},"observation_digest":"sha256:5684ea67b2f1563e0b98fe8a1ef58f92849e791606e0538e4208e876b99729c8","observation_id":"358e0bb5-b56f-4ab7-91df-b8ce2596b5a0","resolution":{"observed_at":"2026-05-18T02:05:39.054797Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":"2407.00079","doi":"10.48550/arxiv.2407.00079","metadata_source":"pith","pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mooncake: A kvcache-centric disaggregated architecture for llm serving","venue":"cs.DC","work_id":"eee8f937-db6b-4c48-887b-960ca80c0d81","year":2024},"citing_paper":{"arxiv_id":"2603.21354","last_updated":"2026-04-08T22:31:39Z","snapshot_observed_at":"2026-07-06T22:50:05.914888Z","submitted_at":"2026-03-22T18:30:11Z","title":"The Workload-Router-Pool Architecture for LLM Inference Optimization: A Vision Paper from the vLLM Semantic Router Project","version":2},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-15T06:40:27.945478Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2603.21354"},"observation_digest":"sha256:77d90ba0e5702db5ff778905215d9ec9b8f6d87148745a8bfaa134a036253786","observation_id":"a9981e83-bb93-4c9b-b8c6-f65b2d426fc6","resolution":{"observed_at":"2026-05-15T06:45:12.282525Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":"2407.00079","doi":"10.48550/arxiv.2407.00079","metadata_source":"pith","pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mooncake: A kvcache-centric disaggregated architecture for llm serving","venue":"cs.DC","work_id":"eee8f937-db6b-4c48-887b-960ca80c0d81","year":2024},"citing_paper":{"arxiv_id":"2604.03044","last_updated":"2026-04-08T07:22:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-03T13:52:38Z","title":"JoyAI-LLM Flash: Advancing Mid-Scale LLMs with Token Efficiency","version":2},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-05-13T19:26:38.134505Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2604.03044"},"observation_digest":"sha256:ea678fb53a2cf9160b6cfb8aed180a5810642f4ee9be5943ca77569d568da43b","observation_id":"2c786606-936f-41f7-9486-acba5244a115","resolution":{"observed_at":"2026-05-13T19:28:09.832457Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":"2407.00079","doi":"10.48550/arxiv.2407.00079","metadata_source":"pith","pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mooncake: A kvcache-centric disaggregated architecture for llm serving","venue":"cs.DC","work_id":"eee8f937-db6b-4c48-887b-960ca80c0d81","year":2024},"citing_paper":{"arxiv_id":"2604.16007","last_updated":"2026-04-17T12:29:54Z","snapshot_observed_at":"2026-07-06T23:03:26.308324Z","submitted_at":"2026-04-17T12:29:54Z","title":"MemExplorer: Navigating the Heterogeneous Memory Design Space for Agentic Inference NPUs","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-10T07:45:17.043107Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2604.16007"},"observation_digest":"sha256:6ab5f0167c765118b7ca922c771178986a70ea680db09f98cd2a45b00dcaa6b6","observation_id":"be385161-c75d-4823-ac03-f1f4d621b35d","resolution":{"observed_at":"2026-05-10T07:47:12.427852Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":"2407.00079","doi":"10.48550/arxiv.2407.00079","metadata_source":"pith","pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mooncake: A kvcache-centric disaggregated architecture for llm serving","venue":"cs.DC","work_id":"eee8f937-db6b-4c48-887b-960ca80c0d81","year":2024},"citing_paper":{"arxiv_id":"2604.24820","last_updated":"2026-04-27T14:06:21Z","snapshot_observed_at":"2026-07-31T06:39:38.066325Z","submitted_at":"2026-04-27T14:06:21Z","title":"Salca: A Sparsity-Aware Hardware Accelerator for Efficient Long-Context Attention Decoding","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-07T17:56:39.124969Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2604.24820"},"observation_digest":"sha256:39a32cfab87ca7d3dce8cd114646069dc27fddd5a0df3864110db781e8774731","observation_id":"1324798b-10f5-435d-887e-e830606d2d60","resolution":{"observed_at":"2026-05-11T23:11:18.872688Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":"2407.00079","doi":"10.48550/arxiv.2407.00079","metadata_source":"pith","pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mooncake: A kvcache-centric disaggregated architecture for llm serving","venue":"cs.DC","work_id":"eee8f937-db6b-4c48-887b-960ca80c0d81","year":2024},"citing_paper":{"arxiv_id":"2604.26968","last_updated":"2026-04-19T21:34:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-19T21:34:09Z","title":"Predictive Multi-Tier Memory Management for KV Cache in Large-Scale GPU Inference","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-10T05:01:02.726796Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2604.26968"},"observation_digest":"sha256:82feda7e12fb0a6bc8d9f43b871d73631081e1b65aa36e7fe9e110fee3f05132","observation_id":"ce4ab281-2187-4221-b962-1aa18009bdbf","resolution":{"observed_at":"2026-05-10T10:14:10.662004Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":"2407.00079","doi":"10.48550/arxiv.2407.00079","metadata_source":"pith","pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mooncake: A kvcache-centric disaggregated architecture for llm serving","venue":"cs.DC","work_id":"eee8f937-db6b-4c48-887b-960ca80c0d81","year":2024},"citing_paper":{"arxiv_id":"2605.03375","last_updated":"2026-05-05T05:33:11Z","snapshot_observed_at":"2026-07-06T23:16:20.322265Z","submitted_at":"2026-05-05T05:33:11Z","title":"Tutti: Making SSD-Backed KV Cache Practical for Long-Context LLM Serving","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-09T16:19:33.613685Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2605.03375"},"observation_digest":"sha256:8efeec6fbdde3103368563ac6c5ed5159037087492bdc9bfe1b73cdc180a3954","observation_id":"72ce4e52-8721-4738-a326-af124bf4f46b","resolution":{"observed_at":"2026-05-11T16:31:10.413581Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":"2407.00079","doi":"10.48550/arxiv.2407.00079","metadata_source":"pith","pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mooncake: A kvcache-centric disaggregated architecture for llm serving","venue":"cs.DC","work_id":"eee8f937-db6b-4c48-887b-960ca80c0d81","year":2024},"citing_paper":{"arxiv_id":"2605.14217","last_updated":"2026-05-14T00:19:41Z","snapshot_observed_at":"2026-07-06T23:25:39.415637Z","submitted_at":"2026-05-14T00:19:41Z","title":"PreFT: Prefill-only finetuning for efficient inference","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-05-15T02:10:14.721584Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2605.14217"},"observation_digest":"sha256:5a62b2bc6b12d3e8abcef003aa13c5cf3da5eed73b79a902fcdc9e83f63b2d20","observation_id":"0c86b6ee-87f1-4c9d-9069-54b6c96603dc","resolution":{"observed_at":"2026-05-15T02:13:30.631370Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":"2407.00079","doi":"10.48550/arxiv.2407.00079","metadata_source":"pith","pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mooncake: A kvcache-centric disaggregated architecture for llm serving","venue":"cs.DC","work_id":"eee8f937-db6b-4c48-887b-960ca80c0d81","year":2024},"citing_paper":{"arxiv_id":"2606.00288","last_updated":"2026-06-24T07:36:07Z","snapshot_observed_at":"2026-07-06T23:41:01.066075Z","submitted_at":"2026-05-29T19:20:16Z","title":"Model-Native Computing Architecture: Envisioning Future System Architecture Through the Lens of Computer Architecture","version":3},"reference_index":128,"source":"pdf_text","source_observed_at":"2026-06-28T22:12:13.114405Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2606.00288"},"observation_digest":"sha256:548e6398724adcdd20a7340f242fafbce490767933ae0084df93bcfe1ef68fe6","observation_id":"f6e62323-3a28-47cb-85f4-a532990b95b5","resolution":{"observed_at":"2026-07-01T19:36:09.325179Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":"2407.00079","doi":"10.48550/arxiv.2407.00079","metadata_source":"pith","pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mooncake: A kvcache-centric disaggregated architecture for llm serving","venue":"cs.DC","work_id":"eee8f937-db6b-4c48-887b-960ca80c0d81","year":2024},"citing_paper":{"arxiv_id":"2606.00866","last_updated":"2026-05-30T19:44:25Z","snapshot_observed_at":"2026-08-06T18:53:41.253747Z","submitted_at":"2026-05-30T19:44:25Z","title":"Idleness is Relative: Exploiting Tool-Call Idle Windows for Offloading in Agentic Systems with MORI","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-06-28T17:30:56.324289Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2606.00866"},"observation_digest":"sha256:347a1ec6b839ccf6e4bda1a33be77299b13d6bf76a308c34b4cd4aa77cb4f13b","observation_id":"87eee698-7063-4fc6-a4ec-335ac8a4eaf7","resolution":{"observed_at":"2026-07-01T21:06:13.602230Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":"2407.00079","doi":"10.48550/arxiv.2407.00079","metadata_source":"pith","pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mooncake: A kvcache-centric disaggregated architecture for llm serving","venue":"cs.DC","work_id":"eee8f937-db6b-4c48-887b-960ca80c0d81","year":2024},"citing_paper":{"arxiv_id":"2606.06453","last_updated":"2026-06-04T17:48:17Z","snapshot_observed_at":"2026-07-06T23:46:15.936608Z","submitted_at":"2026-06-04T17:48:17Z","title":"Vortex: Efficient and Programmable Sparse Attention Serving for AI Agents","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-06-28T01:07:14.691347Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2606.06453"},"observation_digest":"sha256:15a1c8940f03081a94439bc374e975b8086cc95467a35e47cff762476a155893","observation_id":"da99d0d7-d3d8-4dc5-96d0-80e2439e0af0","resolution":{"observed_at":"2026-07-02T13:36:59.446547Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":"2407.00079","doi":"10.48550/arxiv.2407.00079","metadata_source":"pith","pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mooncake: A kvcache-centric disaggregated architecture for llm serving","venue":"cs.DC","work_id":"eee8f937-db6b-4c48-887b-960ca80c0d81","year":2024},"citing_paper":{"arxiv_id":"2606.07684","last_updated":"2026-06-05T02:46:04Z","snapshot_observed_at":"2026-08-06T15:37:49.045087Z","submitted_at":"2026-06-05T02:46:04Z","title":"Semantic Cache Distillation: Efficient State Transfer via Reuse and Selective Patching","version":1},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-06-27T22:43:25.637631Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2606.07684"},"observation_digest":"sha256:e03673619a491fd08efce5d38052b684ea80a638daa48f7662152f4b5e3d5b31","observation_id":"b99726db-bcc4-451a-a747-db9dd14c3604","resolution":{"observed_at":"2026-07-02T16:27:09.081015Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":"2407.00079","doi":"10.48550/arxiv.2407.00079","metadata_source":"pith","pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mooncake: A kvcache-centric disaggregated architecture for llm serving","venue":"cs.DC","work_id":"eee8f937-db6b-4c48-887b-960ca80c0d81","year":2024},"citing_paper":{"arxiv_id":"2606.08635","last_updated":"2026-06-07T13:57:05Z","snapshot_observed_at":"2026-08-02T07:41:43.943071Z","submitted_at":"2026-06-07T13:57:05Z","title":"SpectrumKV: Per-Token Mixed-Precision KV Cache Transfer for Prefill-Decode Disaggregated LLM Serving","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-06-27T18:50:50.066713Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2606.08635"},"observation_digest":"sha256:7b2e5bbf3b50ad904131acbf7eeb02d715f9f675195a8755b4ff5474e329ee91","observation_id":"6a7026a3-6fad-4535-9b78-648b29be30b2","resolution":{"observed_at":"2026-07-02T22:27:26.298208Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":"2407.00079","doi":"10.48550/arxiv.2407.00079","metadata_source":"pith","pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mooncake: A kvcache-centric disaggregated architecture for llm serving","venue":"cs.DC","work_id":"eee8f937-db6b-4c48-887b-960ca80c0d81","year":2024},"citing_paper":{"arxiv_id":"2606.11690","last_updated":"2026-06-10T06:07:24Z","snapshot_observed_at":"2026-08-03T02:14:00.113361Z","submitted_at":"2026-06-10T06:07:24Z","title":"Beyond Per-Token Pricing: A Concurrency-Aware Methodology for LLM Infrastructure Cost Estimation","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-06-27T08:45:42.781160Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2606.11690"},"observation_digest":"sha256:2108194981d61f7141547db9e78653e5d301f3599777a9284feda16c977edd9c","observation_id":"e304f344-56ad-4675-bbe0-9395700334cb","resolution":{"observed_at":"2026-07-03T12:48:11.785527Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":"2407.00079","doi":"10.48550/arxiv.2407.00079","metadata_source":"pith","pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mooncake: A kvcache-centric disaggregated architecture for llm serving","venue":"cs.DC","work_id":"eee8f937-db6b-4c48-887b-960ca80c0d81","year":2024},"citing_paper":{"arxiv_id":"2606.20295","last_updated":"2026-07-24T07:54:17Z","snapshot_observed_at":"2026-08-02T10:48:55.986116Z","submitted_at":"2026-06-18T14:33:09Z","title":"Token-Operations-Oriented Inference Optimization Techniques for Large Models","version":1},"reference_index":176,"source":"pdf_text","source_observed_at":"2026-06-26T16:15:22.543601Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2606.20295"},"observation_digest":"sha256:7dcb2ce3a5c7d242a7d9214ca4451d88fd00e45c5fca372089ab4ad3a40f382e","observation_id":"ef0ee4aa-1e77-4571-83ad-aee61d130932","resolution":{"observed_at":"2026-07-04T05:09:36.750859Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-02T10:49:19.644055Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.20295","last_updated":"2026-07-24T07:54:17Z","snapshot_observed_at":"2026-08-02T10:48:55.986116Z","submitted_at":"2026-06-18T14:33:09Z","title":"Token-Operations-Oriented Inference Optimization Techniques for Large Models","version":2},"reference_index":176,"source":"pdf_text","source_observed_at":"2026-08-02T10:49:19.644055Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2606.20295"},"observation_digest":"sha256:f2815b2db851aa7c2b1381020e03055c2940f6d29b628bb93dd457f3fa8d75b0","observation_id":"a1bed497-9edd-4dc1-a395-b959a7092a3d","resolution":{"observed_at":"2026-08-02T10:49:19.644055Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":"2407.00079","doi":"10.48550/arxiv.2407.00079","metadata_source":"pith","pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mooncake: A kvcache-centric disaggregated architecture for llm serving","venue":"cs.DC","work_id":"eee8f937-db6b-4c48-887b-960ca80c0d81","year":2024},"citing_paper":{"arxiv_id":"2606.20577","last_updated":"2026-05-03T05:31:35Z","snapshot_observed_at":"2026-08-06T09:06:29.399648Z","submitted_at":"2026-05-03T05:31:35Z","title":"Human-Less LLM Serving: Quantifying the Human Tax on Throughput","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-07-01T00:44:55.517266Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2606.20577"},"observation_digest":"sha256:1b401987d37c88b6cfe70199f1a9c8f0270fd61f73f29e545b24d3c741050d0e","observation_id":"96d4f8d0-d408-420f-aa10-6104605c5c63","resolution":{"observed_at":"2026-07-01T00:45:11.798434Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":"2407.00079","doi":"10.48550/arxiv.2407.00079","metadata_source":"pith","pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mooncake: A kvcache-centric disaggregated architecture for llm serving","venue":"cs.DC","work_id":"eee8f937-db6b-4c48-887b-960ca80c0d81","year":2024},"citing_paper":{"arxiv_id":"2606.29708","last_updated":"2026-06-30T03:13:37Z","snapshot_observed_at":"2026-07-07T00:03:41.292432Z","submitted_at":"2026-06-29T02:24:13Z","title":"Demystifying the Design Space and Best Practices for Heterogeneous LLM Inference and Serving","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-30T05:37:13.211613Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2606.29708"},"observation_digest":"sha256:585773a2c4af894e6feca916ee50c741e201dc08f06a5f8fd14a6203543518f5","observation_id":"b1e392b9-f9d0-46e6-ac5c-b3c2d2f25b59","resolution":{"observed_at":"2026-06-30T14:04:45.338293Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":"2407.00079","doi":"10.48550/arxiv.2407.00079","metadata_source":"pith","pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mooncake: A kvcache-centric disaggregated architecture for llm serving","venue":"cs.DC","work_id":"eee8f937-db6b-4c48-887b-960ca80c0d81","year":2024},"citing_paper":{"arxiv_id":"2606.29708","last_updated":"2026-06-30T03:13:37Z","snapshot_observed_at":"2026-07-07T00:03:41.292432Z","submitted_at":"2026-06-29T02:24:13Z","title":"Demystifying the Design Space and Best Practices for Heterogeneous LLM Inference and Serving","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-07-01T07:06:53.318182Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2606.29708"},"observation_digest":"sha256:b8970ac5e809166fa385dda38169ac01518ea92195af539f5e1802741ccca7b6","observation_id":"1d063ebc-92c8-4e23-a621-98eefecb3671","resolution":{"observed_at":"2026-07-01T08:55:35.140206Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":"2407.00079","doi":"10.48550/arxiv.2407.00079","metadata_source":"pith","pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mooncake: A kvcache-centric disaggregated architecture for llm serving","venue":"cs.DC","work_id":"eee8f937-db6b-4c48-887b-960ca80c0d81","year":2024},"citing_paper":{"arxiv_id":"2606.31093","last_updated":"2026-06-30T03:29:21Z","snapshot_observed_at":"2026-08-01T17:26:20.214245Z","submitted_at":"2026-06-30T03:29:21Z","title":"Omni-Flow: A Unified Workflow Orchestration and Distributed KV Cache Sharing Framework for Multimodal Inference","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-07-01T04:13:53.722933Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2606.31093"},"observation_digest":"sha256:62d4df40302194e9cc5340e2364624db80958ecc2b62aaf69870731dbc948224","observation_id":"50f5c434-9115-4a2f-b9d4-92fe427608c8","resolution":{"observed_at":"2026-07-01T11:35:43.612928Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":"2407.00079","doi":"10.48550/arxiv.2407.00079","metadata_source":"pith","pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mooncake: A kvcache-centric disaggregated architecture for llm serving","venue":"cs.DC","work_id":"eee8f937-db6b-4c48-887b-960ca80c0d81","year":2024},"citing_paper":{"arxiv_id":"2607.00466","last_updated":"2026-07-02T08:02:42Z","snapshot_observed_at":"2026-08-02T22:04:11.416989Z","submitted_at":"2026-07-01T05:34:38Z","title":"ELDR: Expert-Locality-Aware Decode Routing for PD-Disaggregated MoE Serving","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-07-02T06:40:53.505889Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2607.00466"},"observation_digest":"sha256:7fb7d22fc71e6332d58f92b2ec60e32da1448cea44afa0f1ce7f6f6862834fcd","observation_id":"f2898e7b-bcdb-45b4-8c5e-0560faa4ad5a","resolution":{"observed_at":"2026-07-02T06:56:43.946753Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":"2407.00079","doi":"10.48550/arxiv.2407.00079","metadata_source":"pith","pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mooncake: A kvcache-centric disaggregated architecture for llm serving","venue":"cs.DC","work_id":"eee8f937-db6b-4c48-887b-960ca80c0d81","year":2024},"citing_paper":{"arxiv_id":"2607.00466","last_updated":"2026-07-02T08:02:42Z","snapshot_observed_at":"2026-08-02T22:04:11.416989Z","submitted_at":"2026-07-01T05:34:38Z","title":"ELDR: Expert-Locality-Aware Decode Routing for PD-Disaggregated MoE Serving","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-07-03T18:58:44.830766Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2607.00466"},"observation_digest":"sha256:b8322c556611e561dea3cdff3532e1530e64f60d34e6b2ac50fc097a5483000e","observation_id":"69825baa-86d1-4130-823e-eadfc4c44395","resolution":{"observed_at":"2026-07-03T18:58:50.500809Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":"2407.00079","doi":"10.48550/arxiv.2407.00079","metadata_source":"pith","pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mooncake: A kvcache-centric disaggregated architecture for llm serving","venue":"cs.DC","work_id":"eee8f937-db6b-4c48-887b-960ca80c0d81","year":2024},"citing_paper":{"arxiv_id":"2607.01579","last_updated":"2026-07-02T01:23:02Z","snapshot_observed_at":"2026-08-02T15:54:32.616830Z","submitted_at":"2026-07-02T01:23:02Z","title":"OmniPilot: An Uncertainty-Aware LLM Inference Advisor for Heterogeneous GPU Clusters","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-07-03T06:30:30.308713Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2607.01579"},"observation_digest":"sha256:e7a3cd33699e035f1d2952bef26ce49b472a7f1d9f39609f5b25f2441c0152ab","observation_id":"36bbb44a-af48-4c3d-9700-7d0f15b55344","resolution":{"observed_at":"2026-07-03T06:37:42.198474Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-07-11T21:08:24.159706Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.04181","last_updated":"2026-07-05T08:52:41Z","snapshot_observed_at":"2026-08-06T05:24:59.303832Z","submitted_at":"2026-07-05T08:52:41Z","title":"CoCoScale: Leveraging Layer-wise Scaling to Unlock the Potential of Online LLM Serving","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-07-11T21:08:24.159706Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2607.04181"},"observation_digest":"sha256:d6393a120415a591d81463e3f65e8467367bd356621d5a790891cef97469a85f","observation_id":"d8b16788-b522-4b7e-ab51-38379c76484a","resolution":{"observed_at":"2026-07-11T21:08:24.159706Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":"2407.00079","doi":"10.48550/arxiv.2407.00079","metadata_source":"pith","pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mooncake: A kvcache-centric disaggregated architecture for llm serving","venue":"cs.DC","work_id":"eee8f937-db6b-4c48-887b-960ca80c0d81","year":2024},"citing_paper":{"arxiv_id":"2607.05876","last_updated":"2026-07-08T02:47:41Z","snapshot_observed_at":"2026-08-05T07:46:13.093303Z","submitted_at":"2026-07-07T06:11:54Z","title":"Think Before You Grid-Search: Floor-First Triage for LLM Serving","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-07-08T22:38:12.637901Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2607.05876"},"observation_digest":"sha256:5dcf05241e169669fe9d18b2a41e8debae178a14de9834724de3e962b3962c58","observation_id":"b019a78b-e277-413a-8f33-392ccf9cb36d","resolution":{"observed_at":"2026-07-08T22:45:40.129136Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":"2407.00079","doi":"10.48550/arxiv.2407.00079","metadata_source":"pith","pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mooncake: A kvcache-centric disaggregated architecture for llm serving","venue":"cs.DC","work_id":"eee8f937-db6b-4c48-887b-960ca80c0d81","year":2024},"citing_paper":{"arxiv_id":"2607.05876","last_updated":"2026-07-08T02:47:41Z","snapshot_observed_at":"2026-08-05T07:46:13.093303Z","submitted_at":"2026-07-07T06:11:54Z","title":"Think Before You Grid-Search: Floor-First Triage for LLM Serving","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-07-11T01:55:09.658053Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2607.05876"},"observation_digest":"sha256:138664bd844a44a0aec938753385181c8c8592b7ea46f513e0580ecde1cd0f86","observation_id":"bdbfc038-faf2-431a-afd8-67b23139ec73","resolution":{"observed_at":"2026-07-11T01:57:51.505587Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":"2407.00079","doi":"10.48550/arxiv.2407.00079","metadata_source":"pith","pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mooncake: A kvcache-centric disaggregated architecture for llm serving","venue":"cs.DC","work_id":"eee8f937-db6b-4c48-887b-960ca80c0d81","year":2024},"citing_paper":{"arxiv_id":"2607.06519","last_updated":"2026-07-07T17:26:28Z","snapshot_observed_at":"2026-08-06T13:47:36.380110Z","submitted_at":"2026-07-07T17:26:28Z","title":"FreqDepthKV: Frequency-Guided Depth Sharing for Robust KV Cache Compression in Long-Context LLM Inference","version":1},"reference_index":103,"source":"arxiv_source","source_observed_at":"2026-07-11T00:12:13.830917Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2607.06519"},"observation_digest":"sha256:e2a0258b448170e90b7e96fd86dc8e1196b518289ebfab292e85add606a47def","observation_id":"d36af72a-b138-47a3-8cc6-7be560f40b68","resolution":{"observed_at":"2026-07-11T00:17:45.637015Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":"2407.00079","doi":"10.48550/arxiv.2407.00079","metadata_source":"pith","pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mooncake: A kvcache-centric disaggregated architecture for llm serving","venue":"cs.DC","work_id":"eee8f937-db6b-4c48-887b-960ca80c0d81","year":2024},"citing_paper":{"arxiv_id":"2607.06523","last_updated":"2026-07-07T17:29:01Z","snapshot_observed_at":"2026-07-10T23:18:28.096764Z","submitted_at":"2026-07-07T17:29:01Z","title":"DepthWeave-KV: Token-Adaptive Cross-Layer Residual Factorization for Long-Context KV Cache Compression","version":1},"reference_index":117,"source":"arxiv_source","source_observed_at":"2026-07-08T03:07:23.648382Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2607.06523"},"observation_digest":"sha256:eb8cdf935b465d641c08d3169e7d1bbc995283ed1b8932eb6466895ee1b970d0","observation_id":"6a9d0331-1118-4278-9fc4-43b1397b7ba6","resolution":{"observed_at":"2026-07-08T03:14:31.493985Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":"2407.00079","doi":"10.48550/arxiv.2407.00079","metadata_source":"pith","pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mooncake: A kvcache-centric disaggregated architecture for llm serving","venue":"cs.DC","work_id":"eee8f937-db6b-4c48-887b-960ca80c0d81","year":2024},"citing_paper":{"arxiv_id":"2607.08032","last_updated":"2026-07-09T01:15:03Z","snapshot_observed_at":"2026-08-02T16:38:21.894642Z","submitted_at":"2026-07-09T01:15:03Z","title":"What to Keep, What to Forget: A Rate--Distortion View of Memory Compaction in LLMs and Agents","version":1},"reference_index":96,"source":"pdf_text","source_observed_at":"2026-07-10T01:26:59.421158Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2607.08032"},"observation_digest":"sha256:b1135f371f7fc02fd0d93bf0e7bdc79f8dce19aa820fc53bd9807e0e7ffb179c","observation_id":"b2f41ac6-904f-47b6-9e19-4a5677ee4d3c","resolution":{"observed_at":"2026-07-10T01:36:44.325524Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":"2407.00079","doi":"10.48550/arxiv.2407.00079","metadata_source":"pith","pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mooncake: A kvcache-centric disaggregated architecture for llm serving","venue":"cs.DC","work_id":"eee8f937-db6b-4c48-887b-960ca80c0d81","year":2024},"citing_paper":{"arxiv_id":"2607.08057","last_updated":"2026-07-09T02:11:18Z","snapshot_observed_at":"2026-08-04T11:12:32.355642Z","submitted_at":"2026-07-09T02:11:18Z","title":"Towards Efficient Large Language Model Serving: A Survey on System-Aware KV Cache Optimization","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-07-10T00:55:52.215700Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2607.08057"},"observation_digest":"sha256:1f7427c4b4ee64f085b1d6e124b647c6d450a919fcbfdb4815218229c595d1b8","observation_id":"43c055a7-a8eb-4daf-8c8d-f31ee9ecee5d","resolution":{"observed_at":"2026-07-10T00:56:40.960364Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T18:20:11.013176+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-07-13T07:32:31.866525Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.08786","last_updated":"2026-06-13T13:38:27Z","snapshot_observed_at":"2026-07-15T23:17:52.421187Z","submitted_at":"2026-06-13T13:38:27Z","title":"Accelerating GPU Inference of Large Language Models with Moderately Unstructured Sparse Weight Matrices","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-07-13T07:32:31.866525Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2607.08786"},"observation_digest":"sha256:214a19cc8cbb77a8550f891ab0ed13924b8e0f087968b57a4800965b172420eb","observation_id":"2f6ad3a4-c991-4d54-9e5d-9ccff413a265","resolution":{"observed_at":"2026-07-13T07:32:31.866525Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-03T00:55:34.428229Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.28633","last_updated":"2026-04-19T22:46:57Z","snapshot_observed_at":"2026-08-05T23:11:21.305292Z","submitted_at":"2026-04-19T22:46:57Z","title":"Topology-Aware Data Movement for Disaggregated GPU Inference","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-03T00:55:34.428229Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2607.28633"},"observation_digest":"sha256:b3d3b1e7d9e9cd5c986e7872fec0c1cf72bd8792384485be022d2276c4a70552","observation_id":"1e5e8596-d161-4b47-ab91-067c04b95cb9","resolution":{"observed_at":"2026-08-03T00:55:34.428229Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-06T00:33:26.100297Z","title":"Mooncake: A KVCache-centric disaggregated architecture for LLM serving,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.01126","last_updated":"2026-08-02T09:51:15Z","snapshot_observed_at":"2026-08-06T23:25:38.408965Z","submitted_at":"2026-08-02T09:51:15Z","title":"Spatial Prefix Caching for Wireless Edge LLM Inference: A Stochastic-Geometry and Queueing Framework","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T00:33:26.100297Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2608.01126"},"observation_digest":"sha256:2d601dbb82555060f355c392b60fdc20459e8a3afeaefcc507cc0be0a36cb135","observation_id":"8ed69102-1abf-4a82-988c-ccbeda51369b","resolution":{"observed_at":"2026-08-06T00:33:26.100297Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2407.00079/citation-record","integrity":"/paper/2407.00079/integrity","json":"/paper/2407.00079/citation-record.json","paper":"/paper/2407.00079"},"outbound":[],"paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","latest_version":4,"primary_category":"cs.DC","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 45 inbound Pith citation observations for arXiv:2407.00079."}