{"as_of":"2026-08-07T18:00:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:85969b2a4113387cf4b5619644e85e68bb3b6f5ec1c69d00c7b7f1b35a81e165","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":23,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":23,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":23,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":23,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T10:26:32.736179Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":6,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2401.11181","last_updated":"2024-01-20T09:43:36Z","snapshot_observed_at":"2026-08-05T23:35:16.350557Z","submitted_at":"2024-01-20T09:43:36Z","title":"Inference without Interference: Disaggregate LLM Inference for Mixed Downstream Workloads","version":1},"cited_work":{"arxiv_id":"2401.11181","doi":"10.48550/arxiv.2401.11181","metadata_source":"arxiv_reference","pith_arxiv_id":"2401.11181","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"2024-01-20","venue":"arXiv (Cornell University)","work_id":"6cbae121-60a0-46b6-8c38-52dcd1e105c2","year":2024},"citing_paper":{"arxiv_id":"2404.14294","last_updated":"2024-07-19T04:47:36Z","snapshot_observed_at":"2026-07-06T18:03:47.096406Z","submitted_at":"2024-04-22T15:53:08Z","title":"A Survey on Efficient Inference for Large Language Models","version":3},"reference_index":273,"source":"pdf_text","source_observed_at":"2026-05-15T02:39:33.007894Z"},"links":{"cited_paper":"/paper/2401.11181","citing_paper":"/paper/2404.14294"},"observation_digest":"sha256:defd7ce00d3ab70999392f33f11b0611a9514022daa430c8ab84ec53ba136342","observation_id":"cd8da1e9-bc32-448d-bd9d-4a7ffbce1000","resolution":{"observed_at":"2026-05-15T02:39:33.469145Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.11181","last_updated":"2024-01-20T09:43:36Z","snapshot_observed_at":"2026-08-05T23:35:16.350557Z","submitted_at":"2024-01-20T09:43:36Z","title":"Inference without Interference: Disaggregate LLM Inference for Mixed Downstream Workloads","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.11181","snapshot_observed_at":"2026-08-07T10:26:32.736179Z","title":"Inference without Inter- ference: Disaggregate LLM Inference for Mixed Downstream Workloads","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05508","last_updated":"2025-06-05T18:47:49Z","snapshot_observed_at":"2026-08-07T13:07:21.433278Z","submitted_at":"2025-06-05T18:47:49Z","title":"Beyond the Buzz: A Pragmatic Take on Inference Disaggregation","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T10:26:32.736179Z"},"links":{"cited_paper":"/paper/2401.11181","citing_paper":"/paper/2506.05508"},"observation_digest":"sha256:8ec248636e6d4e68285d30d289a43aaae97fb7563571645cc31c7777d7324a7c","observation_id":"d539e8c3-45f1-43ce-a2b0-9593364440c7","resolution":{"observed_at":"2026-08-07T10:26:32.736179Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.11181","last_updated":"2024-01-20T09:43:36Z","snapshot_observed_at":"2026-08-05T23:35:16.350557Z","submitted_at":"2024-01-20T09:43:36Z","title":"Inference without Interference: Disaggregate LLM Inference for Mixed Downstream Workloads","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.11181","snapshot_observed_at":"2026-08-06T19:03:53.051327Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.06608","last_updated":"2025-08-07T12:26:15Z","snapshot_observed_at":"2026-08-06T18:56:52.854557Z","submitted_at":"2025-07-09T07:27:18Z","title":"Nexus:Proactive Intra-GPU Disaggregation of Prefill and Decode in LLM Serving","version":5},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:53.051327Z"},"links":{"cited_paper":"/paper/2401.11181","citing_paper":"/paper/2507.06608"},"observation_digest":"sha256:7428f299af2edfdc1b8d4828f0c2e8d82ecf4b92efd085791382b439195372a2","observation_id":"5f148349-ce14-4b64-8106-9106552df42e","resolution":{"observed_at":"2026-08-06T19:03:53.051327Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.11181","last_updated":"2024-01-20T09:43:36Z","snapshot_observed_at":"2026-08-05T23:35:16.350557Z","submitted_at":"2024-01-20T09:43:36Z","title":"Inference without Interference: Disaggregate LLM Inference for Mixed Downstream Workloads","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.11181","snapshot_observed_at":"2026-08-05T15:45:36.345043Z","title":"Infer- ence without interference: Disaggregate llm inference for mixed downstream workloads","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.19559","last_updated":"2025-08-27T04:22:02Z","snapshot_observed_at":"2026-08-06T12:06:42.525601Z","submitted_at":"2025-08-27T04:22:02Z","title":"Taming the Chaos: Coordinated Autoscaling for Heterogeneous and Disaggregated LLM Inference","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-05T15:45:36.345043Z"},"links":{"cited_paper":"/paper/2401.11181","citing_paper":"/paper/2508.19559"},"observation_digest":"sha256:6c2c3dabd8db2a10be65e2a91a026999cbcd6c078f6e8a11d08bd5a720c83097","observation_id":"f33f8277-8b6a-4991-b6f4-b43908aa318b","resolution":{"observed_at":"2026-08-05T15:45:36.345043Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.11181","last_updated":"2024-01-20T09:43:36Z","snapshot_observed_at":"2026-08-05T23:35:16.350557Z","submitted_at":"2024-01-20T09:43:36Z","title":"Inference without Interference: Disaggregate LLM Inference for Mixed Downstream Workloads","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.11181","snapshot_observed_at":"2026-08-04T20:53:31.939770Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.08309","last_updated":"2025-09-10T06:06:51Z","snapshot_observed_at":"2026-08-06T05:24:56.655402Z","submitted_at":"2025-09-10T06:06:51Z","title":"Hetis: Serving LLMs in Heterogeneous GPU Clusters with Fine-grained and Dynamic Parallelism","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-04T20:53:31.939770Z"},"links":{"cited_paper":"/paper/2401.11181","citing_paper":"/paper/2509.08309"},"observation_digest":"sha256:d8faae2b5e54bcf98c94400ab1378346a8af24421c411e19b89bca4962fc815a","observation_id":"1f96ef1a-0a70-4f5b-96ed-3cd4aff33490","resolution":{"observed_at":"2026-08-04T20:53:31.939770Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.11181","last_updated":"2024-01-20T09:43:36Z","snapshot_observed_at":"2026-08-05T23:35:16.350557Z","submitted_at":"2024-01-20T09:43:36Z","title":"Inference without Interference: Disaggregate LLM Inference for Mixed Downstream Workloads","version":1},"cited_work":{"arxiv_id":"2401.11181","doi":"10.48550/arxiv.2401.11181","metadata_source":"arxiv_reference","pith_arxiv_id":"2401.11181","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"2024-01-20","venue":"arXiv (Cornell University)","work_id":"6cbae121-60a0-46b6-8c38-52dcd1e105c2","year":2024},"citing_paper":{"arxiv_id":"2510.11938","last_updated":"2026-05-09T02:54:37Z","snapshot_observed_at":"2026-07-06T22:32:32.632779Z","submitted_at":"2025-10-13T21:01:40Z","title":"FlexPipe: Adapting Dynamic LLM Serving Through Inflight Pipeline Refactoring in Fragmented Serverless Clusters","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-18T07:07:46.228491Z"},"links":{"cited_paper":"/paper/2401.11181","citing_paper":"/paper/2510.11938"},"observation_digest":"sha256:e3d88fb15db5f70e0b63b21a92b2533fff5558631c7d008527bcf62a011690e2","observation_id":"26ad8cfa-5076-4b7b-b8bd-67eb2edd7fbc","resolution":{"observed_at":"2026-05-18T07:11:03.617983Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.11181","last_updated":"2024-01-20T09:43:36Z","snapshot_observed_at":"2026-08-05T23:35:16.350557Z","submitted_at":"2024-01-20T09:43:36Z","title":"Inference without Interference: Disaggregate LLM Inference for Mixed Downstream Workloads","version":1},"cited_work":{"arxiv_id":"2401.11181","doi":"10.48550/arxiv.2401.11181","metadata_source":"arxiv_reference","pith_arxiv_id":"2401.11181","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"2024-01-20","venue":"arXiv (Cornell University)","work_id":"6cbae121-60a0-46b6-8c38-52dcd1e105c2","year":2024},"citing_paper":{"arxiv_id":"2510.13668","last_updated":"2026-05-04T10:30:41Z","snapshot_observed_at":"2026-07-30T08:04:44.387106Z","submitted_at":"2025-10-15T15:29:08Z","title":"STAR: Decode-Phase Rescheduling for LLM Inference","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-18T06:11:28.120860Z"},"links":{"cited_paper":"/paper/2401.11181","citing_paper":"/paper/2510.13668"},"observation_digest":"sha256:93c1067d019cc6c8db48ee0ace262d2c4bb0e47952b72b002f8baf2242ad8327","observation_id":"1a795543-9eae-427b-8f11-15ae5f4ad8c2","resolution":{"observed_at":"2026-05-18T06:12:25.938366Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.11181","last_updated":"2024-01-20T09:43:36Z","snapshot_observed_at":"2026-08-05T23:35:16.350557Z","submitted_at":"2024-01-20T09:43:36Z","title":"Inference without Interference: Disaggregate LLM Inference for Mixed Downstream Workloads","version":1},"cited_work":{"arxiv_id":"2401.11181","doi":"10.48550/arxiv.2401.11181","metadata_source":"arxiv_reference","pith_arxiv_id":"2401.11181","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"2024-01-20","venue":"arXiv (Cornell University)","work_id":"6cbae121-60a0-46b6-8c38-52dcd1e105c2","year":2024},"citing_paper":{"arxiv_id":"2512.09427","last_updated":"2026-04-21T07:27:04Z","snapshot_observed_at":"2026-07-06T22:38:35.387503Z","submitted_at":"2025-12-10T08:52:20Z","title":"ODMA: On-Demand Memory Allocation Strategy for LLM Serving on LPDDR-Class Accelerators","version":5},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-16T23:54:08.052482Z"},"links":{"cited_paper":"/paper/2401.11181","citing_paper":"/paper/2512.09427"},"observation_digest":"sha256:35c5cb6fa946f03b7744061e6130cbd06a46d4fd25be11db060b71a5e0b19582","observation_id":"0af25593-fd88-4ed5-954b-b2bef949192e","resolution":{"observed_at":"2026-05-16T23:58:43.005876Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.11181","last_updated":"2024-01-20T09:43:36Z","snapshot_observed_at":"2026-08-05T23:35:16.350557Z","submitted_at":"2024-01-20T09:43:36Z","title":"Inference without Interference: Disaggregate LLM Inference for Mixed Downstream Workloads","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.11181","snapshot_observed_at":"2026-08-02T23:16:44.283021Z","title":"Inference without interference: Disaggregate llm inference for mixed down- stream workloads.arXiv preprint arXiv:2401.11181,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.14516","last_updated":"2026-07-21T08:53:51Z","snapshot_observed_at":"2026-08-04T10:17:52.933347Z","submitted_at":"2026-02-16T07:07:30Z","title":"Efficient Multi-round LLM Inference over Disaggregated Serving","version":2},"reference_index":1994,"source":"pdf_text","source_observed_at":"2026-08-02T23:16:44.283021Z"},"links":{"cited_paper":"/paper/2401.11181","citing_paper":"/paper/2602.14516"},"observation_digest":"sha256:b908e1e7e63cd50ab16c0db4619a4c6f02fadd35f9c270b2544c8eb9f60b07da","observation_id":"d20fdeaf-85f2-4752-af64-7f07217c58e6","resolution":{"observed_at":"2026-08-02T23:16:44.283021Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.11181","last_updated":"2024-01-20T09:43:36Z","snapshot_observed_at":"2026-08-05T23:35:16.350557Z","submitted_at":"2024-01-20T09:43:36Z","title":"Inference without Interference: Disaggregate LLM Inference for Mixed Downstream Workloads","version":1},"cited_work":{"arxiv_id":"2401.11181","doi":"10.48550/arxiv.2401.11181","metadata_source":"arxiv_reference","pith_arxiv_id":"2401.11181","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"2024-01-20","venue":"arXiv (Cornell University)","work_id":"6cbae121-60a0-46b6-8c38-52dcd1e105c2","year":2024},"citing_paper":{"arxiv_id":"2605.04357","last_updated":"2026-05-05T23:25:29Z","snapshot_observed_at":"2026-07-06T23:17:09.246351Z","submitted_at":"2026-05-05T23:25:29Z","title":"Coral: Cost-Efficient Multi-LLM Serving over Heterogeneous Cloud GPUs","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-08T16:45:39.853764Z"},"links":{"cited_paper":"/paper/2401.11181","citing_paper":"/paper/2605.04357"},"observation_digest":"sha256:9dfbfee199448e09793e3efc7abbb152bbb39aedaeeb8aed1f209b63b8623cdb","observation_id":"1ce18c33-284a-4fde-b632-ee06b6f0368e","resolution":{"observed_at":"2026-05-11T18:01:07.014104Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.11181","last_updated":"2024-01-20T09:43:36Z","snapshot_observed_at":"2026-08-05T23:35:16.350557Z","submitted_at":"2024-01-20T09:43:36Z","title":"Inference without Interference: Disaggregate LLM Inference for Mixed Downstream Workloads","version":1},"cited_work":{"arxiv_id":"2401.11181","doi":"10.48550/arxiv.2401.11181","metadata_source":"arxiv_reference","pith_arxiv_id":"2401.11181","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"2024-01-20","venue":"arXiv (Cornell University)","work_id":"6cbae121-60a0-46b6-8c38-52dcd1e105c2","year":2024},"citing_paper":{"arxiv_id":"2605.17613","last_updated":"2026-05-17T19:18:39Z","snapshot_observed_at":"2026-07-06T23:28:36.134614Z","submitted_at":"2026-05-17T19:18:39Z","title":"VeriCache: Turning Lossy KV Cache into Lossless LLM Inference","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-19T22:09:15.782104Z"},"links":{"cited_paper":"/paper/2401.11181","citing_paper":"/paper/2605.17613"},"observation_digest":"sha256:416fe924cd35ad550621f4f6dee21193135f3bb10613ba1abd8d43a50518e913","observation_id":"193c6fde-95d1-4e3c-bdc9-5cfc75a66375","resolution":{"observed_at":"2026-05-19T22:12:51.167065Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.11181","last_updated":"2024-01-20T09:43:36Z","snapshot_observed_at":"2026-08-05T23:35:16.350557Z","submitted_at":"2024-01-20T09:43:36Z","title":"Inference without Interference: Disaggregate LLM Inference for Mixed Downstream Workloads","version":1},"cited_work":{"arxiv_id":"2401.11181","doi":"10.48550/arxiv.2401.11181","metadata_source":"arxiv_reference","pith_arxiv_id":"2401.11181","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"2024-01-20","venue":"arXiv (Cornell University)","work_id":"6cbae121-60a0-46b6-8c38-52dcd1e105c2","year":2024},"citing_paper":{"arxiv_id":"2605.23389","last_updated":"2026-05-22T09:00:45Z","snapshot_observed_at":"2026-08-01T23:40:06.222566Z","submitted_at":"2026-05-22T09:00:45Z","title":"AlignedServe: Orchestrating Prefix-aware Batching to Build a High-throughput and Computing-efficient LLM Serving System","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-25T03:12:49.028342Z"},"links":{"cited_paper":"/paper/2401.11181","citing_paper":"/paper/2605.23389"},"observation_digest":"sha256:146df158ea7a0aacff83439746240ede729d130c3cdcedad435fbb7af36309fe","observation_id":"93fbeaed-f880-4c22-88d8-a296d194c9da","resolution":{"observed_at":"2026-05-25T03:15:17.288802Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.11181","last_updated":"2024-01-20T09:43:36Z","snapshot_observed_at":"2026-08-05T23:35:16.350557Z","submitted_at":"2024-01-20T09:43:36Z","title":"Inference without Interference: Disaggregate LLM Inference for Mixed Downstream Workloads","version":1},"cited_work":{"arxiv_id":"2401.11181","doi":"10.48550/arxiv.2401.11181","metadata_source":"arxiv_reference","pith_arxiv_id":"2401.11181","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"2024-01-20","venue":"arXiv (Cornell University)","work_id":"6cbae121-60a0-46b6-8c38-52dcd1e105c2","year":2024},"citing_paper":{"arxiv_id":"2606.20577","last_updated":"2026-05-03T05:31:35Z","snapshot_observed_at":"2026-08-06T09:06:29.399648Z","submitted_at":"2026-05-03T05:31:35Z","title":"Human-Less LLM Serving: Quantifying the Human Tax on Throughput","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-07-01T00:44:55.517266Z"},"links":{"cited_paper":"/paper/2401.11181","citing_paper":"/paper/2606.20577"},"observation_digest":"sha256:85bbe822a8ad9266c0a4b7db3225e71780d62c81a892ea8f201c2493ab5bcb70","observation_id":"34fb4efc-43e1-4098-ae80-ff6ca91927d7","resolution":{"observed_at":"2026-07-01T00:45:11.801306Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.11181","last_updated":"2024-01-20T09:43:36Z","snapshot_observed_at":"2026-08-05T23:35:16.350557Z","submitted_at":"2024-01-20T09:43:36Z","title":"Inference without Interference: Disaggregate LLM Inference for Mixed Downstream Workloads","version":1},"cited_work":{"arxiv_id":"2401.11181","doi":"10.48550/arxiv.2401.11181","metadata_source":"arxiv_reference","pith_arxiv_id":"2401.11181","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"2024-01-20","venue":"arXiv (Cornell University)","work_id":"6cbae121-60a0-46b6-8c38-52dcd1e105c2","year":2024},"citing_paper":{"arxiv_id":"2606.22327","last_updated":"2026-06-24T14:02:53Z","snapshot_observed_at":"2026-08-07T17:40:04.855669Z","submitted_at":"2026-06-21T04:05:38Z","title":"Geometry-Aware Online Scheduling for LLM Serving: From Theoretical Bound to System Practice","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-06-26T11:17:35.860286Z"},"links":{"cited_paper":"/paper/2401.11181","citing_paper":"/paper/2606.22327"},"observation_digest":"sha256:c4f30d0ed2b42b1065b35eb6e8e29c04df59f580cf09fd7e860541c064df388a","observation_id":"66c5f74e-e132-435a-8a50-d755ca4e86e2","resolution":{"observed_at":"2026-07-04T08:39:41.985966Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.11181","last_updated":"2024-01-20T09:43:36Z","snapshot_observed_at":"2026-08-05T23:35:16.350557Z","submitted_at":"2024-01-20T09:43:36Z","title":"Inference without Interference: Disaggregate LLM Inference for Mixed Downstream Workloads","version":1},"cited_work":{"arxiv_id":"2401.11181","doi":"10.48550/arxiv.2401.11181","metadata_source":"arxiv_reference","pith_arxiv_id":"2401.11181","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"2024-01-20","venue":"arXiv (Cornell University)","work_id":"6cbae121-60a0-46b6-8c38-52dcd1e105c2","year":2024},"citing_paper":{"arxiv_id":"2606.29708","last_updated":"2026-06-30T03:13:37Z","snapshot_observed_at":"2026-07-07T00:03:41.292432Z","submitted_at":"2026-06-29T02:24:13Z","title":"Demystifying the Design Space and Best Practices for Heterogeneous LLM Inference and Serving","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-30T05:37:13.211613Z"},"links":{"cited_paper":"/paper/2401.11181","citing_paper":"/paper/2606.29708"},"observation_digest":"sha256:119b4eade36684510f211181a76df44a236d8050021913f411ef9b03d2dcc5a2","observation_id":"91619bb5-87eb-472e-8b08-6cf731256fcb","resolution":{"observed_at":"2026-06-30T14:04:45.341416Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.11181","last_updated":"2024-01-20T09:43:36Z","snapshot_observed_at":"2026-08-05T23:35:16.350557Z","submitted_at":"2024-01-20T09:43:36Z","title":"Inference without Interference: Disaggregate LLM Inference for Mixed Downstream Workloads","version":1},"cited_work":{"arxiv_id":"2401.11181","doi":"10.48550/arxiv.2401.11181","metadata_source":"arxiv_reference","pith_arxiv_id":"2401.11181","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"2024-01-20","venue":"arXiv (Cornell University)","work_id":"6cbae121-60a0-46b6-8c38-52dcd1e105c2","year":2024},"citing_paper":{"arxiv_id":"2606.29708","last_updated":"2026-06-30T03:13:37Z","snapshot_observed_at":"2026-07-07T00:03:41.292432Z","submitted_at":"2026-06-29T02:24:13Z","title":"Demystifying the Design Space and Best Practices for Heterogeneous LLM Inference and Serving","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-07-01T07:06:53.318182Z"},"links":{"cited_paper":"/paper/2401.11181","citing_paper":"/paper/2606.29708"},"observation_digest":"sha256:f93de88399458d9539af12745f5656d14bc095c8872efeadb30809abd6320aa0","observation_id":"04af9445-3e67-4eb8-b5ea-8d59b1fe8549","resolution":{"observed_at":"2026-07-01T08:55:35.143870Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.11181","last_updated":"2024-01-20T09:43:36Z","snapshot_observed_at":"2026-08-05T23:35:16.350557Z","submitted_at":"2024-01-20T09:43:36Z","title":"Inference without Interference: Disaggregate LLM Inference for Mixed Downstream Workloads","version":1},"cited_work":{"arxiv_id":"2401.11181","doi":"10.48550/arxiv.2401.11181","metadata_source":"arxiv_reference","pith_arxiv_id":"2401.11181","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"2024-01-20","venue":"arXiv (Cornell University)","work_id":"6cbae121-60a0-46b6-8c38-52dcd1e105c2","year":2024},"citing_paper":{"arxiv_id":"2607.02043","last_updated":"2026-07-02T11:10:05Z","snapshot_observed_at":"2026-07-07T00:07:33.184524Z","submitted_at":"2026-07-02T11:10:05Z","title":"Towards Load-Aware Prefill Deflection for Disaggregated LLM Serving","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-07-03T06:05:06.467649Z"},"links":{"cited_paper":"/paper/2401.11181","citing_paper":"/paper/2607.02043"},"observation_digest":"sha256:a76b2d407abd2c55166cbdf255820bbd997ade84b6a00b84e5cee4eeb9550284","observation_id":"9045a234-d428-4ed8-947a-61dad41f26e5","resolution":{"observed_at":"2026-07-03T06:07:40.890885Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.11181","last_updated":"2024-01-20T09:43:36Z","snapshot_observed_at":"2026-08-05T23:35:16.350557Z","submitted_at":"2024-01-20T09:43:36Z","title":"Inference without Interference: Disaggregate LLM Inference for Mixed Downstream Workloads","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.11181","snapshot_observed_at":"2026-07-11T21:08:24.159706Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.04181","last_updated":"2026-07-05T08:52:41Z","snapshot_observed_at":"2026-08-06T05:24:59.303832Z","submitted_at":"2026-07-05T08:52:41Z","title":"CoCoScale: Leveraging Layer-wise Scaling to Unlock the Potential of Online LLM Serving","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-07-11T21:08:24.159706Z"},"links":{"cited_paper":"/paper/2401.11181","citing_paper":"/paper/2607.04181"},"observation_digest":"sha256:b3691ca6be4635b4265d14a97a3bb97b53c6259b480e08421dbaf9cec3479276","observation_id":"ac815cd7-9513-4cb1-b790-b1cb655a2309","resolution":{"observed_at":"2026-07-11T21:08:24.159706Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.11181","last_updated":"2024-01-20T09:43:36Z","snapshot_observed_at":"2026-08-05T23:35:16.350557Z","submitted_at":"2024-01-20T09:43:36Z","title":"Inference without Interference: Disaggregate LLM Inference for Mixed Downstream Workloads","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.11181","snapshot_observed_at":"2026-07-11T20:58:04.182764Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.04206","last_updated":"2026-07-05T09:55:29Z","snapshot_observed_at":"2026-08-05T08:10:46.508162Z","submitted_at":"2026-07-05T09:55:29Z","title":"Sangam: Efficiently Serving Diffusion LLMs with the AR Stack","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-07-11T20:58:04.182764Z"},"links":{"cited_paper":"/paper/2401.11181","citing_paper":"/paper/2607.04206"},"observation_digest":"sha256:8945bb39fdc53c404366cf876fc552cb3005b9b0755c234ba267009132445651","observation_id":"ebab7f57-b3e2-4028-9376-3b53766a92b9","resolution":{"observed_at":"2026-07-11T20:58:04.182764Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.11181","last_updated":"2024-01-20T09:43:36Z","snapshot_observed_at":"2026-08-05T23:35:16.350557Z","submitted_at":"2024-01-20T09:43:36Z","title":"Inference without Interference: Disaggregate LLM Inference for Mixed Downstream Workloads","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.11181","snapshot_observed_at":"2026-07-14T03:18:47.408836Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.11770","last_updated":"2026-07-13T16:28:41Z","snapshot_observed_at":"2026-08-06T14:29:01.585049Z","submitted_at":"2026-07-13T16:28:41Z","title":"AutoSLO: Practical Latency SLOs on Cloud Data Warehouses -- Extended Version","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-07-14T03:18:47.408836Z"},"links":{"cited_paper":"/paper/2401.11181","citing_paper":"/paper/2607.11770"},"observation_digest":"sha256:31b3965046b66e9256acae19d7ea92488c712c9f71811ef507eb8712098fcbce","observation_id":"25516814-d561-4c3a-aa0a-af61bd4072d0","resolution":{"observed_at":"2026-07-14T03:18:47.408836Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.11181","last_updated":"2024-01-20T09:43:36Z","snapshot_observed_at":"2026-08-05T23:35:16.350557Z","submitted_at":"2024-01-20T09:43:36Z","title":"Inference without Interference: Disaggregate LLM Inference for Mixed Downstream Workloads","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.11181","snapshot_observed_at":"2026-07-31T16:21:14.862617Z","title":"Inference without interference: Disaggregate LLM in- ference for mixed downstream workloads.CoRR, abs/2401.11181, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.28150","last_updated":"2026-07-30T12:56:05Z","snapshot_observed_at":"2026-08-06T19:16:15.787392Z","submitted_at":"2026-07-30T12:56:05Z","title":"SmartGen: Seamless Disaggregated LLM Inference with Selective KV Cache Transfer","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-07-31T16:21:14.862617Z"},"links":{"cited_paper":"/paper/2401.11181","citing_paper":"/paper/2607.28150"},"observation_digest":"sha256:db327048d4a31beb38ed992ba841939bc96a6155b08c848c3c9c1e253c8e7aa0","observation_id":"88f9298f-b660-4f3d-8b42-61c58cd035e9","resolution":{"observed_at":"2026-07-31T16:21:14.862617Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.11181","last_updated":"2024-01-20T09:43:36Z","snapshot_observed_at":"2026-08-05T23:35:16.350557Z","submitted_at":"2024-01-20T09:43:36Z","title":"Inference without Interference: Disaggregate LLM Inference for Mixed Downstream Workloads","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.11181","snapshot_observed_at":"2026-08-03T14:14:10.457557Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.29069","last_updated":"2026-07-31T06:44:06Z","snapshot_observed_at":"2026-08-07T03:44:55.840492Z","submitted_at":"2026-07-31T06:44:06Z","title":"Rethinking AI Cloud Infrastructure for Agentic Serving Systems with the Aries Experimentation Framework","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-03T14:14:10.457557Z"},"links":{"cited_paper":"/paper/2401.11181","citing_paper":"/paper/2607.29069"},"observation_digest":"sha256:c3e92771ceeb5bf948a4b9553fe120d6aebbc72223b34febeeb946d0cc814411","observation_id":"b49bd378-5512-4d7b-b8f6-08a197292084","resolution":{"observed_at":"2026-08-03T14:14:10.457557Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.11181","last_updated":"2024-01-20T09:43:36Z","snapshot_observed_at":"2026-08-05T23:35:16.350557Z","submitted_at":"2024-01-20T09:43:36Z","title":"Inference without Interference: Disaggregate LLM Inference for Mixed Downstream Workloads","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.11181","snapshot_observed_at":"2026-08-04T18:49:22.951059Z","title":"Inference without interference: Disaggre- gate llm inference for mixed downstream workloads,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.01891","last_updated":"2026-08-03T08:32:50Z","snapshot_observed_at":"2026-08-06T23:33:24.457506Z","submitted_at":"2026-08-03T08:32:50Z","title":"Energy-Efficient LLM Serving via Disaggregated Attention--FFN and Flexible Frequency Scaling","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-04T18:49:22.951059Z"},"links":{"cited_paper":"/paper/2401.11181","citing_paper":"/paper/2608.01891"},"observation_digest":"sha256:bbd105747f79fe80dab13e1c2ab899a4e09d752ebe15b647a534128b5e399a5a","observation_id":"7d2fc877-c3e9-4a5b-9681-8eb28e071b5f","resolution":{"observed_at":"2026-08-04T18:49:22.951059Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2401.11181/citation-record","integrity":"/paper/2401.11181/integrity","json":"/paper/2401.11181/citation-record.json","paper":"/paper/2401.11181"},"outbound":[],"paper":{"arxiv_id":"2401.11181","last_updated":"2024-01-20T09:43:36Z","latest_version":1,"primary_category":"cs.DC","snapshot_observed_at":"2026-08-05T23:35:16.350557Z","submitted_at":"2024-01-20T09:43:36Z","title":"Inference without Interference: Disaggregate LLM Inference for Mixed Downstream Workloads"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 23 inbound Pith citation observations for arXiv:2401.11181."}