{"as_of":"2026-08-17T13:18:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:4b333a725a496e55c5edb6b04033b6393f70d0d4450232cab111047a87f28396","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":12,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":12,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-17T06:30:58.91139+00:00","state":"measured"},{"denominator":12,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":12,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T05:49:09.890421Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-03T04:37:36.387611Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2408.08147","last_updated":"2024-08-15T13:32:25Z","snapshot_observed_at":"2026-08-16T13:26:15.478706Z","submitted_at":"2024-08-15T13:32:25Z","title":"P/D-Serve: Serving Disaggregated Large Language Model at Scale","version":1},"cited_work":{"arxiv_id":"2408.08147","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.08147","snapshot_observed_at":"2026-07-03T04:37:36.387611Z","title":"P/d-serve: Serving disaggregated large language model at scale,","venue":null,"work_id":"1e46ffa7-bd2a-410b-a63d-32b2c2094fb2","year":2024},"citing_paper":{"arxiv_id":"2412.03594","last_updated":"2026-04-22T15:33:51Z","snapshot_observed_at":"2026-08-15T02:09:13.690958Z","submitted_at":"2024-11-29T05:57:37Z","title":"BatchLLM: Optimizing Large Batched LLM Inference with Global Prefix Sharing and Throughput-oriented Token Batching","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-23T16:57:46.645061Z"},"links":{"cited_paper":"/paper/2408.08147","citing_paper":"/paper/2412.03594"},"observation_digest":"sha256:cac2eb796e296dabc37527ef67c4751a7779991d7ea218a5f5d0478c7f5559d4","observation_id":"0f66a650-246b-4c2b-aa0d-2e1ff2812c67","resolution":{"observed_at":"2026-05-23T16:58:11.931791Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.08147","last_updated":"2024-08-15T13:32:25Z","snapshot_observed_at":"2026-08-16T13:26:15.478706Z","submitted_at":"2024-08-15T13:32:25Z","title":"P/D-Serve: Serving Disaggregated Large Language Model at Scale","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.08147","snapshot_observed_at":"2026-08-11T14:05:01.802775Z","title":"Kwon, W., Li, Z., Zhuang, S., Sheng, Y ., Zheng, L., Yu, C","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.12488","last_updated":"2024-12-17T02:44:43Z","snapshot_observed_at":"2026-08-14T19:21:18.649075Z","submitted_at":"2024-12-17T02:44:43Z","title":"A System for Microserving of LLMs","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-11T14:05:01.802775Z"},"links":{"cited_paper":"/paper/2408.08147","citing_paper":"/paper/2412.12488"},"observation_digest":"sha256:7cb4478d67318bd3f81b3a6c0c9daea6f9188f303aa129d0520900abc5f39722","observation_id":"73cd2cb3-f961-43ff-bf15-056782726c6c","resolution":{"observed_at":"2026-08-11T14:05:01.802775Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.08147","last_updated":"2024-08-15T13:32:25Z","snapshot_observed_at":"2026-08-16T13:26:15.478706Z","submitted_at":"2024-08-15T13:32:25Z","title":"P/D-Serve: Serving Disaggregated Large Language Model at Scale","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.08147","snapshot_observed_at":"2026-08-11T05:05:22.508469Z","title":"P/d-serve: Serving disag- gregated large language model at scale","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.18106","last_updated":"2024-12-24T02:27:44Z","snapshot_observed_at":"2026-08-16T13:18:23.068773Z","submitted_at":"2024-12-24T02:27:44Z","title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-11T05:05:22.508469Z"},"links":{"cited_paper":"/paper/2408.08147","citing_paper":"/paper/2412.18106"},"observation_digest":"sha256:642698d116cffc7686388cf2c2f86427d6cd90acd5e907a9671948b6c84aa928","observation_id":"9857202e-31fb-4f66-baee-4a8f4072702d","resolution":{"observed_at":"2026-08-11T05:05:22.508469Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.08147","last_updated":"2024-08-15T13:32:25Z","snapshot_observed_at":"2026-08-16T13:26:15.478706Z","submitted_at":"2024-08-15T13:32:25Z","title":"P/D-Serve: Serving Disaggregated Large Language Model at Scale","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.08147","snapshot_observed_at":"2026-08-11T04:29:34.722701Z","title":"P/d-serve: Serving disaggregated large language model at scale","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.05460","last_updated":"2025-06-28T03:53:17Z","snapshot_observed_at":"2026-08-17T08:53:57.735754Z","submitted_at":"2024-12-25T10:11:31Z","title":"Efficiently Serving Large Multimodal Models Using EPD Disaggregation","version":4},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-11T04:29:34.722701Z"},"links":{"cited_paper":"/paper/2408.08147","citing_paper":"/paper/2501.05460"},"observation_digest":"sha256:8f75c03f00250f56bd73245a8784b2dd77b64d83d24486be547e5eb3eef4cda8","observation_id":"a0117dca-f92c-49d7-9e15-5433c53e1873","resolution":{"observed_at":"2026-08-11T04:29:34.722701Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.08147","last_updated":"2024-08-15T13:32:25Z","snapshot_observed_at":"2026-08-16T13:26:15.478706Z","submitted_at":"2024-08-15T13:32:25Z","title":"P/D-Serve: Serving Disaggregated Large Language Model at Scale","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.08147","snapshot_observed_at":"2026-08-08T11:32:21.519527Z","title":"P/d-serve: Serving disaggregated large language model at scale","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.07903","last_updated":"2025-02-11T19:17:35Z","snapshot_observed_at":"2026-08-14T00:16:40.934984Z","submitted_at":"2025-02-11T19:17:35Z","title":"HexGen-2: Disaggregated Generative Inference of LLMs in Heterogeneous Environment","version":1},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-08T11:32:21.519527Z"},"links":{"cited_paper":"/paper/2408.08147","citing_paper":"/paper/2502.07903"},"observation_digest":"sha256:7d109bcf0d01caf8135a9c80570c4bf513e07ed79e8c5c70bc59dcd2fc943b83","observation_id":"9669077d-a821-4144-b871-35e595bbefee","resolution":{"observed_at":"2026-08-08T11:32:21.519527Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.08147","last_updated":"2024-08-15T13:32:25Z","snapshot_observed_at":"2026-08-16T13:26:15.478706Z","submitted_at":"2024-08-15T13:32:25Z","title":"P/D-Serve: Serving Disaggregated Large Language Model at Scale","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.08147","snapshot_observed_at":"2026-08-08T10:10:22.915120Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.08182","last_updated":"2025-02-12T07:42:45Z","snapshot_observed_at":"2026-08-17T11:57:14.492319Z","submitted_at":"2025-02-12T07:42:45Z","title":"Memory Offloading for Large Language Model Inference with Latency SLO Guarantees","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-08T10:10:22.915120Z"},"links":{"cited_paper":"/paper/2408.08147","citing_paper":"/paper/2502.08182"},"observation_digest":"sha256:32b36563e5110cd3c07f1958740a2589a3ca874b01480562f5a334fa685d4955","observation_id":"98edf3ec-67d3-4150-bca6-d89a37b78ddb","resolution":{"observed_at":"2026-08-08T10:10:22.915120Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.08147","last_updated":"2024-08-15T13:32:25Z","snapshot_observed_at":"2026-08-16T13:26:15.478706Z","submitted_at":"2024-08-15T13:32:25Z","title":"P/D-Serve: Serving Disaggregated Large Language Model at Scale","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.08147","snapshot_observed_at":"2026-08-16T05:49:09.890421Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.19720","last_updated":"2025-04-28T12:14:02Z","snapshot_observed_at":"2026-08-16T07:43:12.291655Z","submitted_at":"2025-04-28T12:14:02Z","title":"Taming the Titans: A Survey of Efficient LLM Inference Serving","version":1},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-08-16T05:49:09.890421Z"},"links":{"cited_paper":"/paper/2408.08147","citing_paper":"/paper/2504.19720"},"observation_digest":"sha256:45e482d8e04dc54949947b3f532e70a608860dd8324e87ab343db13f7b028bb8","observation_id":"47c8d098-b2b8-41c6-b079-79c7f3870010","resolution":{"observed_at":"2026-08-16T05:49:09.890421Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.08147","last_updated":"2024-08-15T13:32:25Z","snapshot_observed_at":"2026-08-16T13:26:15.478706Z","submitted_at":"2024-08-15T13:32:25Z","title":"P/D-Serve: Serving Disaggregated Large Language Model at Scale","version":1},"cited_work":{"arxiv_id":"2408.08147","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.08147","snapshot_observed_at":"2026-07-03T04:37:36.387611Z","title":"P/d-serve: Serving disaggregated large language model at scale,","venue":null,"work_id":"1e46ffa7-bd2a-410b-a63d-32b2c2094fb2","year":2024},"citing_paper":{"arxiv_id":"2505.09999","last_updated":"2026-05-11T03:41:00Z","snapshot_observed_at":"2026-07-30T05:05:51.180006Z","submitted_at":"2025-05-15T06:24:08Z","title":"ServeGen: Workload Characterization and Generation of Large Language Model Serving in Production","version":3},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-22T15:42:05.266854Z"},"links":{"cited_paper":"/paper/2408.08147","citing_paper":"/paper/2505.09999"},"observation_digest":"sha256:fef314c9dabc7fa02c0ecf02789d2a7f16a2b162d3682dad7473febe6eeead7a","observation_id":"927d0b0a-ff32-4d29-b92a-b6fe002c767d","resolution":{"observed_at":"2026-05-22T15:44:58.145493Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.08147","last_updated":"2024-08-15T13:32:25Z","snapshot_observed_at":"2026-08-16T13:26:15.478706Z","submitted_at":"2024-08-15T13:32:25Z","title":"P/D-Serve: Serving Disaggregated Large Language Model at Scale","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.08147","snapshot_observed_at":"2026-08-07T10:26:32.597438Z","title":"P/D-Serve: Serving Disaggregated Large Language Model at Scale","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05508","last_updated":"2025-06-05T18:47:49Z","snapshot_observed_at":"2026-08-15T06:01:30.055870Z","submitted_at":"2025-06-05T18:47:49Z","title":"Beyond the Buzz: A Pragmatic Take on Inference Disaggregation","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T10:26:32.597438Z"},"links":{"cited_paper":"/paper/2408.08147","citing_paper":"/paper/2506.05508"},"observation_digest":"sha256:6ddbb9f62f3634b3580fcf4767c523fdab70f9dfaf80b10172af96fa86b85852","observation_id":"8e2c26b6-936b-41dc-b964-713400ed1a7d","resolution":{"observed_at":"2026-08-07T10:26:32.597438Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.08147","last_updated":"2024-08-15T13:32:25Z","snapshot_observed_at":"2026-08-16T13:26:15.478706Z","submitted_at":"2024-08-15T13:32:25Z","title":"P/D-Serve: Serving Disaggregated Large Language Model at Scale","version":1},"cited_work":{"arxiv_id":"2408.08147","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.08147","snapshot_observed_at":"2026-07-03T04:37:36.387611Z","title":"P/d-serve: Serving disaggregated large language model at scale,","venue":null,"work_id":"1e46ffa7-bd2a-410b-a63d-32b2c2094fb2","year":2024},"citing_paper":{"arxiv_id":"2607.01617","last_updated":"2026-07-02T02:32:44Z","snapshot_observed_at":"2026-08-02T09:43:01.556074Z","submitted_at":"2026-07-02T02:32:44Z","title":"3DLS: A 3D Logic-Stacked Architecture for Disaggregated LLM Serving","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-07-03T04:29:56.401898Z"},"links":{"cited_paper":"/paper/2408.08147","citing_paper":"/paper/2607.01617"},"observation_digest":"sha256:630a74f23de511d00ab627d3ecdbd05ecd1e73ad17afa570a9fa72092b88fa8a","observation_id":"310aaf6a-68ed-41e1-a343-8c5a5695b27f","resolution":{"observed_at":"2026-07-03T04:37:36.389208Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.08147","last_updated":"2024-08-15T13:32:25Z","snapshot_observed_at":"2026-08-16T13:26:15.478706Z","submitted_at":"2024-08-15T13:32:25Z","title":"P/D-Serve: Serving Disaggregated Large Language Model at Scale","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.08147","snapshot_observed_at":"2026-07-12T09:50:23.266920Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.02574","last_updated":"2026-06-30T16:12:40Z","snapshot_observed_at":"2026-08-13T16:31:58.188253Z","submitted_at":"2026-06-30T16:12:40Z","title":"From Tensor Buffer to Distributed Memory Hierarchy: A Survey of KV Cache Management for LLM Serving","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-07-12T09:50:23.266920Z"},"links":{"cited_paper":"/paper/2408.08147","citing_paper":"/paper/2607.02574"},"observation_digest":"sha256:464ca6bf515685b4cb593dc13f2ed33a163296bc056a5578162763f8702bfd99","observation_id":"08698169-f112-45b2-a260-1d555cd15ccd","resolution":{"observed_at":"2026-07-12T09:50:23.266920Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.08147","last_updated":"2024-08-15T13:32:25Z","snapshot_observed_at":"2026-08-16T13:26:15.478706Z","submitted_at":"2024-08-15T13:32:25Z","title":"P/D-Serve: Serving Disaggregated Large Language Model at Scale","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.08147","snapshot_observed_at":"2026-08-04T20:58:29.272650Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.01785","last_updated":"2026-08-03T06:59:15Z","snapshot_observed_at":"2026-08-15T19:40:12.543019Z","submitted_at":"2026-08-03T06:59:15Z","title":"HorizonServe: Coordinating Request Scheduling with GPU Sharing for Omni-Model Serving","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-04T20:58:29.272650Z"},"links":{"cited_paper":"/paper/2408.08147","citing_paper":"/paper/2608.01785"},"observation_digest":"sha256:9d9a1d536167e5ab72d24b8e4be63b0b98285b0faf09b9e2612e3e1d344ebe83","observation_id":"4f3ce9b1-5086-48fd-93b2-002d9156e69f","resolution":{"observed_at":"2026-08-04T20:58:29.272650Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2408.08147/citation-record","integrity":"/paper/2408.08147/integrity","json":"/paper/2408.08147/citation-record.json","paper":"/paper/2408.08147"},"outbound":[],"paper":{"arxiv_id":"2408.08147","last_updated":"2024-08-15T13:32:25Z","latest_version":1,"primary_category":"cs.DC","snapshot_observed_at":"2026-08-16T13:26:15.478706Z","submitted_at":"2024-08-15T13:32:25Z","title":"P/D-Serve: Serving Disaggregated Large Language Model at Scale"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"thesis":"As of 17 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 12 inbound Pith citation observations for arXiv:2408.08147."}