{"as_of":"2026-08-05T17:13:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:3f086e8da5aa509d596b75f92e611faca2006f622664e8c1d0138c062789b614","coverage":[{"denominator":48,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":48,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-16T06:31:47.330226Z","state":"measured"},{"denominator":109,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":109,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-05T06:32:48.257954+00:00","state":"measured"},{"denominator":61,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":61,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T13:30:38.109614Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"pith","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":15,"observed_at":"2026-08-05T02:28:24.338817Z","source":"pith"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2404.14294","last_updated":"2024-07-19T04:47:36Z","snapshot_observed_at":"2026-07-06T18:03:47.096406Z","submitted_at":"2024-04-22T15:53:08Z","title":"A Survey on Efficient Inference for Large Language Models","version":3},"reference_index":282,"source":"pdf_text","source_observed_at":"2026-05-15T02:39:33.007894Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2404.14294"},"observation_digest":"sha256:c2d60ad48c8bfd37672a7fa64ba4b541b18b1c90205e1316cd939e388126d4a9","observation_id":"fe6461ac-5e7b-4f5d-bc39-59168f834719","resolution":{"observed_at":"2026-05-16T06:31:47.535686Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2409.19256","last_updated":"2024-10-02T04:01:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-28T06:20:03Z","title":"HybridFlow: A Flexible and Efficient RLHF Framework","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-11T07:53:38.715353Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2409.19256"},"observation_digest":"sha256:18552536f3f1a1bf2afb0a7c967f0bd4bfc9b9ecca401e51fc3ae19dbb6a25fc","observation_id":"4abe331b-dea7-429e-ba62-883728ee7af3","resolution":{"observed_at":"2026-05-16T06:31:47.535686Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2410.10819","last_updated":"2024-10-14T17:59:58Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-14T17:59:58Z","title":"DuoAttention: Efficient Long-Context LLM Inference with Retrieval and Streaming Heads","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-05-18T11:49:16.654836Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2410.10819"},"observation_digest":"sha256:4a7723f819102516bab0ce841ac57c8ad069f7465e771f2415a786e318ca71f1","observation_id":"03427540-7973-4bf9-81b7-c76b81325c5a","resolution":{"observed_at":"2026-05-18T11:49:16.707196Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2412.03594","last_updated":"2026-04-22T15:33:51Z","snapshot_observed_at":"2026-07-06T20:01:45.826971Z","submitted_at":"2024-11-29T05:57:37Z","title":"BatchLLM: Optimizing Large Batched LLM Inference with Global Prefix Sharing and Throughput-oriented Token Batching","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-23T16:57:46.645061Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2412.03594"},"observation_digest":"sha256:51c501f83cd53c5c9d5e33ad03c11dcde27c786242c2f4d31b25543b47dbb547","observation_id":"46754dae-cb19-485e-a3fa-b617fed68ded","resolution":{"observed_at":"2026-05-23T16:58:11.909937Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2504.11320","last_updated":"2026-05-14T23:11:43Z","snapshot_observed_at":"2026-08-02T11:51:09.028246Z","submitted_at":"2025-04-15T16:00:21Z","title":"Optimizing LLM Inference: Fluid-Guided Online Scheduling with Memory Constraints","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-22T19:44:35.756018Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2504.11320"},"observation_digest":"sha256:84350f8011c2b917cc9b0f534f52afef0b223afd5ec4e3b5de26fae3a13723e1","observation_id":"69c6f004-7f74-445a-aa37-fe42a90f72e9","resolution":{"observed_at":"2026-05-22T19:45:03.952532Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2505.02922","last_updated":"2026-04-27T10:13:35Z","snapshot_observed_at":"2026-08-02T06:23:14.961990Z","submitted_at":"2025-05-05T18:01:17Z","title":"RetroInfer: A Vector Storage Engine for Scalable Long-Context LLM Inference","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-22T15:59:04.724780Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2505.02922"},"observation_digest":"sha256:4c6a559b7f95a02a097100ee9a87ffee6f6c8d5205150ab8891bf56f47d5e508","observation_id":"2b6323c5-3711-4110-b424-4813dec83e4f","resolution":{"observed_at":"2026-05-16T06:31:47.535686Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T13:30:38.109614Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.00579","last_updated":"2025-08-30T18:25:19Z","snapshot_observed_at":"2026-08-05T13:30:36.860062Z","submitted_at":"2025-08-30T18:25:19Z","title":"KVComp: A High-Performance, LLM-Aware, Lossy Compression Framework for KV Cache","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-05T13:30:38.109614Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2509.00579"},"observation_digest":"sha256:54742b22b0c682ba365667ee6052c2b60a60b73c3711e0049b24feca397c4fc5","observation_id":"9384e789-a628-4784-98f1-570eccf53e47","resolution":{"observed_at":"2026-08-05T13:30:38.109614Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2511.03092","last_updated":"2026-04-08T21:56:44Z","snapshot_observed_at":"2026-08-02T21:39:50.543195Z","submitted_at":"2025-11-05T00:38:31Z","title":"SnapStream: Efficient Long Sequence Decoding on Dataflow Accelerators","version":6},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-18T01:59:30.583027Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2511.03092"},"observation_digest":"sha256:7fefd887ea8d8f8ca85b2efc4c013b54a27c74fe0fb9775012ff1a146eea5f3d","observation_id":"99981bc5-d0a4-49de-a7c1-f551b1b376a8","resolution":{"observed_at":"2026-05-18T02:00:39.372114Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2512.09427","last_updated":"2026-04-21T07:27:04Z","snapshot_observed_at":"2026-07-06T22:38:35.387503Z","submitted_at":"2025-12-10T08:52:20Z","title":"ODMA: On-Demand Memory Allocation Strategy for LLM Serving on LPDDR-Class Accelerators","version":5},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-16T23:54:08.052482Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2512.09427"},"observation_digest":"sha256:175f3b83702b632730d06646c02d0786c515929fba139fd5a474259b90450b19","observation_id":"b847bca7-d887-4b86-a732-357b7bab6771","resolution":{"observed_at":"2026-05-16T23:58:42.971189Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2603.10726","last_updated":"2026-05-20T10:27:28Z","snapshot_observed_at":"2026-07-06T22:48:39.649787Z","submitted_at":"2026-03-11T12:59:12Z","title":"PrefixWall: Mitigating Prefix Caching Side Channels in Shared LLM Systems","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-21T12:14:09.509302Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2603.10726"},"observation_digest":"sha256:f02edead355bdad90c877ed1fe5f04ae2107a966b5fc3b341d25abe9d3cd12a0","observation_id":"477b55fb-b7f4-4549-b48c-4db1c93ae349","resolution":{"observed_at":"2026-05-21T12:15:06.813295Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-02T21:14:46.669096Z","title":"S., and Ramjee, R","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2603.18016","last_updated":"2026-06-01T17:56:47Z","snapshot_observed_at":"2026-08-03T10:44:47.986701Z","submitted_at":"2026-02-24T17:24:50Z","title":"MineDraft: A Framework for Batch Parallel Speculative Decoding","version":2},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-02T21:14:46.669096Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2603.18016"},"observation_digest":"sha256:0e6c14f7ae4c788c9110e1e0581dbe5161bcaa1be7837a13b6d07f2928101cdf","observation_id":"1fe7abec-fc54-4930-b95b-570e5b69389c","resolution":{"observed_at":"2026-08-02T21:14:46.669096Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-07-13T10:30:21.369025Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2604.04251","last_updated":"2026-06-08T14:40:43Z","snapshot_observed_at":"2026-07-13T10:30:20.787060Z","submitted_at":"2026-04-05T20:13:34Z","title":"MC-CPO: Mastery-Conditioned Constrained Policy Optimization for Pedagogically Safe Intelligent Tutoring Systems","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-07-13T10:30:21.369025Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2604.04251"},"observation_digest":"sha256:d44992f8bb203464c04b45443f4dd0c57d73ba0ff3c672b78ff27bda111161f2","observation_id":"6fb2bc04-fe28-4b80-ac28-0de489d8a815","resolution":{"observed_at":"2026-07-13T10:30:21.369025Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2604.07874","last_updated":"2026-04-09T06:45:37Z","snapshot_observed_at":"2026-07-06T22:57:05.146373Z","submitted_at":"2026-04-09T06:45:37Z","title":"Valve: Production Online-Offline Inference Colocation with Jointly-Bounded Preemption Latency and Rate","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T18:05:08.438659Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2604.07874"},"observation_digest":"sha256:e21ff848153ab5fe9d88feb440f89272a8c0636d081da45bc25bf74d8639927a","observation_id":"2095a72b-29ea-4b2a-9214-62c9df2099f3","resolution":{"observed_at":"2026-05-16T06:31:47.535686Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2604.11001","last_updated":"2026-04-13T05:03:16Z","snapshot_observed_at":"2026-07-06T22:59:32.051572Z","submitted_at":"2026-04-13T05:03:16Z","title":"Flow-Controlled Scheduling for LLM Inference with Provable Stability Guarantees","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T15:51:11.343200Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2604.11001"},"observation_digest":"sha256:f2dd52d8a227aa438dd394a4ff3d6f97bf98c89c3be2977d6bbb59972874979b","observation_id":"1d19e4f0-00d1-4c70-adb2-c60c3e58c9b7","resolution":{"observed_at":"2026-05-16T06:31:47.535686Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2604.18788","last_updated":"2026-04-20T19:52:56Z","snapshot_observed_at":"2026-07-31T18:40:43.182032Z","submitted_at":"2026-04-20T19:52:56Z","title":"Efficient Mixture-of-Experts LLM Inference with Apple Silicon NPUs","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-10T05:31:25.826205Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2604.18788"},"observation_digest":"sha256:ced0226de895dff903179188796b117cff526b221a46a06a8841e13a54e915cd","observation_id":"5b5e4f32-6891-40e8-b599-9ed2b0bdcda3","resolution":{"observed_at":"2026-05-16T06:31:47.535686Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2604.24820","last_updated":"2026-04-27T14:06:21Z","snapshot_observed_at":"2026-07-31T06:39:38.066325Z","submitted_at":"2026-04-27T14:06:21Z","title":"Salca: A Sparsity-Aware Hardware Accelerator for Efficient Long-Context Attention Decoding","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-07T17:56:39.124969Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2604.24820"},"observation_digest":"sha256:e9635a0d252fe293ced01e32e2605e77f1a08a3f5057bc1db00ac06b0354f0a4","observation_id":"f552e6cc-291e-4fd6-b5a1-1cb3ebfb1402","resolution":{"observed_at":"2026-05-16T06:31:47.535686Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2604.26103","last_updated":"2026-04-30T09:10:31Z","snapshot_observed_at":"2026-08-04T04:01:51.865779Z","submitted_at":"2026-04-28T20:36:50Z","title":"AMMA: A Multi-Chiplet Memory-Centric Architecture for Low-Latency 1M Context Attention Serving","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-07T14:16:46.241126Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2604.26103"},"observation_digest":"sha256:b8fa3edec8931b19fefe3635f87f6021065bf86587d765f9437a3f1399e17c6e","observation_id":"4c7d2a04-1531-4ce8-bb39-8401e855bb25","resolution":{"observed_at":"2026-05-16T06:31:47.535686Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2604.26837","last_updated":"2026-04-29T16:02:00Z","snapshot_observed_at":"2026-07-06T23:12:24.456023Z","submitted_at":"2026-04-29T16:02:00Z","title":"Unifying Sparse Attention with Hierarchical Memory for Scalable Long-Context LLM Serving","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-07T13:15:21.201950Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2604.26837"},"observation_digest":"sha256:cbb68fc738b7e45a648e709e76d4e3704868273844f7a572667c62301489c61f","observation_id":"57459ce9-f5b2-40e0-8e67-b098b2e4c155","resolution":{"observed_at":"2026-05-16T06:31:47.535686Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2604.26963","last_updated":"2026-04-14T05:15:28Z","snapshot_observed_at":"2026-08-01T00:07:50.693557Z","submitted_at":"2026-04-14T05:15:28Z","title":"MARS: Efficient, Adaptive Co-Scheduling for Heterogeneous Agentic Systems","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-10T14:30:56.899306Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2604.26963"},"observation_digest":"sha256:2d0429588f6cd5b7d51aff2d07a0a8af46e5b66a18683c1a0e1226d62b357826","observation_id":"45f9f4c5-2df1-4438-9432-efc6ee4104f3","resolution":{"observed_at":"2026-05-16T06:31:47.535686Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2604.27476","last_updated":"2026-06-08T07:15:10Z","snapshot_observed_at":"2026-07-06T23:12:57.730833Z","submitted_at":"2026-04-30T06:18:50Z","title":"EdgeFM: Efficient Edge Inference for Vision-Language Models","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-07T08:30:35.264804Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2604.27476"},"observation_digest":"sha256:47f52383bcae605be341996d810a51171558d50bd7063edf9e1234bac34f3e8b","observation_id":"2be23e24-8a6f-4e5d-9e05-c97a6778dd31","resolution":{"observed_at":"2026-05-16T06:31:47.535686Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2604.27476","last_updated":"2026-06-08T07:15:10Z","snapshot_observed_at":"2026-07-06T23:12:57.730833Z","submitted_at":"2026-04-30T06:18:50Z","title":"EdgeFM: Efficient Edge Inference for Vision-Language Models","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-07-01T08:54:41.845820Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2604.27476"},"observation_digest":"sha256:b532b3ca97562069b4022c8d2fbe84bc728d5fe63e214cce2624938788dd2a36","observation_id":"12f4c3e5-f983-4d7e-a3d1-950e5b34a66a","resolution":{"observed_at":"2026-07-01T08:55:34.732042Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2605.00831","last_updated":"2026-03-26T13:27:57Z","snapshot_observed_at":"2026-08-02T06:03:58.648365Z","submitted_at":"2026-03-26T13:27:57Z","title":"GhostServe: A Lightweight Checkpointing System in the Shadow for Fault-Tolerant LLM Serving","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-15T00:37:08.671539Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2605.00831"},"observation_digest":"sha256:307bf1ae10211dfe6e5de28dbee1191f91dd1073755bb551c36899cb809707dc","observation_id":"1f32c318-668d-4544-9bcd-c918306c2685","resolution":{"observed_at":"2026-05-16T06:31:47.535686Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2605.02189","last_updated":"2026-05-04T03:37:40Z","snapshot_observed_at":"2026-08-02T13:10:09.655062Z","submitted_at":"2026-05-04T03:37:40Z","title":"PipeMax: Enhancing Offline LLM Inference on Commodity GPU Servers","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-05-08T18:49:56.357400Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2605.02189"},"observation_digest":"sha256:d877d239b6e89b17260a2ebb9763170bf32c41402608af4be478725877463b23","observation_id":"477ce940-9db6-4f33-af43-ec239089882b","resolution":{"observed_at":"2026-05-16T06:31:47.535686Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2605.02329","last_updated":"2026-05-25T17:26:33Z","snapshot_observed_at":"2026-08-02T00:39:56.914878Z","submitted_at":"2026-05-04T08:29:47Z","title":"Taming Request Imbalance: SLO-Aware Scheduling for Disaggregated LLM Inference","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-08T18:27:13.781231Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2605.02329"},"observation_digest":"sha256:d7f01820755d2a855f02b64120847eabc51cf953a789cae0a7083c5fbfa1cc7c","observation_id":"1963e390-410a-4633-9172-f1778c1d7c3f","resolution":{"observed_at":"2026-05-16T06:31:47.535686Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2605.02329","last_updated":"2026-05-25T17:26:33Z","snapshot_observed_at":"2026-08-02T00:39:56.914878Z","submitted_at":"2026-05-04T08:29:47Z","title":"Taming Request Imbalance: SLO-Aware Scheduling for Disaggregated LLM Inference","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-07-01T00:31:12.965178Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2605.02329"},"observation_digest":"sha256:44c771548554690aa9b6274677024eb450bab6347eb5666b307ea6079f655aba","observation_id":"a5c63d26-c383-43fd-a980-76a81b75c7af","resolution":{"observed_at":"2026-07-01T00:35:10.235272Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2605.02960","last_updated":"2026-05-15T03:29:35Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-05-03T03:10:24Z","title":"MoE-Prefill: Zero Redundancy Overheads in MoE Prefill Serving","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-08T19:29:28.916831Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2605.02960"},"observation_digest":"sha256:72cc3e47a52a1bf9bb34dd1cce9af1891653cc4b32953ee762cb76e3ba470371","observation_id":"e0f9de6d-25c9-47db-8e77-3506007e33a8","resolution":{"observed_at":"2026-05-16T06:31:47.535686Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2605.02960","last_updated":"2026-05-15T03:29:35Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-05-03T03:10:24Z","title":"MoE-Prefill: Zero Redundancy Overheads in MoE Prefill Serving","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-19T17:43:56.767714Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2605.02960"},"observation_digest":"sha256:517330b17b88877308b8851b9fd495a2854ad0f87e03d95d9d84bda03a3d908d","observation_id":"9f64a633-45d3-4dc7-92ba-1a9c74883330","resolution":{"observed_at":"2026-05-19T17:47:42.025621Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2605.08151","last_updated":"2026-05-12T15:30:36Z","snapshot_observed_at":"2026-07-06T23:20:24.341928Z","submitted_at":"2026-05-04T01:27:45Z","title":"SPECTRE: Hybrid Ordinary-Parallel Speculative Serving for Resource-Efficient LLM Inference","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-12T00:56:41.542160Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2605.08151"},"observation_digest":"sha256:4b85a1e4a95fb59079eb8a9acd71f95aee969b1f7e38f69ca1453172e3b9dfaa","observation_id":"294494ee-41a1-4a85-aeab-d53556d90933","resolution":{"observed_at":"2026-05-16T06:31:47.535686Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2605.08151","last_updated":"2026-05-12T15:30:36Z","snapshot_observed_at":"2026-07-06T23:20:24.341928Z","submitted_at":"2026-05-04T01:27:45Z","title":"SPECTRE: Hybrid Ordinary-Parallel Speculative Serving for Resource-Efficient LLM Inference","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-13T07:02:34.494214Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2605.08151"},"observation_digest":"sha256:0bee60b4eed1b58283f83ae5e316966a8f7bd18c7f8bf1962aea32c0e6f341a3","observation_id":"1672a77c-5603-4e69-9190-7b6a1395e35c","resolution":{"observed_at":"2026-05-16T06:31:47.535686Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2605.11744","last_updated":"2026-05-12T08:23:00Z","snapshot_observed_at":"2026-07-06T23:23:30.499023Z","submitted_at":"2026-05-12T08:23:00Z","title":"Training-Inference Consistent Segmented Execution for Long-Context LLMs","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-05-13T06:52:51.852156Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2605.11744"},"observation_digest":"sha256:07d34282eae3362e4812be4e8beb4e5ea98ad97c6365893857495b961d991775","observation_id":"2168dfa0-e6e7-4c2b-beb1-4b679bfe853d","resolution":{"observed_at":"2026-05-16T06:31:47.535686Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2605.11999","last_updated":"2026-05-12T11:48:16Z","snapshot_observed_at":"2026-08-02T19:02:31.798979Z","submitted_at":"2026-05-12T11:48:16Z","title":"The Illusion of Power Capping in LLM Decode: A Phase-Aware Energy Characterisation Across Attention Architectures","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-13T04:48:46.006687Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2605.11999"},"observation_digest":"sha256:ee90f971148f0be96f042388a66974fb75c5fce016de627596715fcf89474fd8","observation_id":"21e964eb-0864-46e1-b933-fce79a301279","resolution":{"observed_at":"2026-05-16T06:31:47.535686Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2605.16839","last_updated":"2026-05-16T06:47:41Z","snapshot_observed_at":"2026-07-06T23:27:57.805018Z","submitted_at":"2026-05-16T06:47:41Z","title":"CompactAttention: Accelerating Chunked Prefill with Block-Union KV Selection","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-19T21:19:31.263068Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2605.16839"},"observation_digest":"sha256:6c5f689814c4bc7333135ac3d33d8572b197460e4ddd66d297edd7651b50f051","observation_id":"fb7488f5-483e-447d-a260-5b81a734ce55","resolution":{"observed_at":"2026-05-19T21:22:47.960862Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2605.18535","last_updated":"2026-05-18T15:18:29Z","snapshot_observed_at":"2026-07-06T23:29:24.494326Z","submitted_at":"2026-05-18T15:18:29Z","title":"Beyond Scaling: Agents Are Heading to the Edge","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-20T11:51:14.999341Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2605.18535"},"observation_digest":"sha256:5afc588ca0fe411ac5e24b7179a18547d670e155e3749970354d49221be3148e","observation_id":"760b1ed1-c8fb-41bc-8920-71fee3715f0e","resolution":{"observed_at":"2026-05-20T11:53:14.935042Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2605.19049","last_updated":"2026-05-18T19:14:52Z","snapshot_observed_at":"2026-07-06T23:29:52.061489Z","submitted_at":"2026-05-18T19:14:52Z","title":"KVBuffer: IO-aware Serving for Linear Attention","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-20T12:00:06.171816Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2605.19049"},"observation_digest":"sha256:38f54804aaa598161a022ba81bc5d86904ae34b7eab76051600235b53f59da51","observation_id":"3aa6fc2c-e0f9-40a7-ba07-69dfe47f7249","resolution":{"observed_at":"2026-05-20T12:03:15.260675Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2605.19593","last_updated":"2026-05-19T09:39:16Z","snapshot_observed_at":"2026-07-06T23:30:21.283826Z","submitted_at":"2026-05-19T09:39:16Z","title":"Towards Multi-Model LLM Schedulers: Empirical Insights into Offloading and Preemption","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-20T05:57:59.624400Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2605.19593"},"observation_digest":"sha256:fadeff026151b3dd25eb336b927b8cde0c2fe66d8d5c240203d9f38b7dca3766","observation_id":"78ccea88-cc51-4bc3-adca-e216a2848772","resolution":{"observed_at":"2026-05-20T05:58:04.862805Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2605.19775","last_updated":"2026-05-19T12:43:51Z","snapshot_observed_at":"2026-08-02T14:22:11.635681Z","submitted_at":"2026-05-19T12:43:51Z","title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-20T02:11:26.925234Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2605.19775"},"observation_digest":"sha256:461973463f5f4a0c1fc7636410a2737de5173898ec07d5530d4277c1bb927e62","observation_id":"2dd0a23f-066c-4141-9f14-2f9b8061721e","resolution":{"observed_at":"2026-05-20T02:12:58.271581Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2605.21312","last_updated":"2026-05-20T15:40:18Z","snapshot_observed_at":"2026-08-02T01:15:29.579250Z","submitted_at":"2026-05-20T15:40:18Z","title":"Frontier: Towards Comprehensive and Accurate LLM Inference Simulation","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-21T03:47:49.835773Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2605.21312"},"observation_digest":"sha256:9934a2294aab37dbb2e257adaadde6a5ce59d11b24863a5d67827ab769ff987d","observation_id":"d1eb6d41-9712-4c4a-bf3a-cb16e797ba18","resolution":{"observed_at":"2026-05-21T03:49:31.093477Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2605.23389","last_updated":"2026-05-22T09:00:45Z","snapshot_observed_at":"2026-08-01T23:40:06.222566Z","submitted_at":"2026-05-22T09:00:45Z","title":"AlignedServe: Orchestrating Prefix-aware Batching to Build a High-throughput and Computing-efficient LLM Serving System","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-25T03:12:49.028342Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2605.23389"},"observation_digest":"sha256:17df356b879bc5a0526f3b4030c1f8ecd9595a9b3da0e342867acd45d8e478ed","observation_id":"4f312682-f682-4d2d-8c83-9582019b93c4","resolution":{"observed_at":"2026-05-25T03:15:17.303901Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2605.24144","last_updated":"2026-05-22T19:06:22Z","snapshot_observed_at":"2026-07-06T23:34:13.859813Z","submitted_at":"2026-05-22T19:06:22Z","title":"EVA: Accelerating LLM Decoding via an Efficient Vector Quantization Architecture","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-30T14:35:02.484377Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2605.24144"},"observation_digest":"sha256:865d8e35259f3bff11a134b66874ef0a9e3c8977b0a727fdbbe3dc3b77778b1d","observation_id":"25b7cead-120f-477d-99f3-08adeda338f8","resolution":{"observed_at":"2026-06-30T14:44:45.607766Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2605.28302","last_updated":"2026-05-27T10:55:57Z","snapshot_observed_at":"2026-08-04T11:39:33.881533Z","submitted_at":"2026-05-27T10:55:57Z","title":"How Far Can Disaggregation Go? A Design-Space Exploration of Attention-FFN Disaggregation for Efficient MoE LLM Serving","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-29T14:11:24.646931Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2605.28302"},"observation_digest":"sha256:081f9a3aa9961fee06f7a551e2c3266cb091be9a5228334ca245478d4d8aef56","observation_id":"eed0d6e2-6f1f-4ac5-abb2-b116cc224555","resolution":{"observed_at":"2026-06-29T14:13:29.987690Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2606.02982","last_updated":"2026-06-19T15:23:40Z","snapshot_observed_at":"2026-07-06T23:43:17.879245Z","submitted_at":"2026-06-02T00:39:31Z","title":"DriftSched: Adaptive QoS-Aware Scheduling under Runtime Token Drift for Multi-Tenant GPU Inference","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-28T07:47:26.074448Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2606.02982"},"observation_digest":"sha256:7bd07e91118455536642e66c8930ac3eb46a4a07ddb92ef3444c5c62688c5d78","observation_id":"748ada5c-771b-459a-a0ab-0341f53d841b","resolution":{"observed_at":"2026-07-02T06:06:40.708633Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2606.05933","last_updated":"2026-06-04T09:36:40Z","snapshot_observed_at":"2026-07-06T23:45:51.910263Z","submitted_at":"2026-06-04T09:36:40Z","title":"Beyond Greedy Chunking: SLO-Aware Sliding-Window Scheduling for LLM Inference","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-27T23:49:28.318260Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2606.05933"},"observation_digest":"sha256:0f7582b0ba4c5cd35df2b2c64172ba93095af71c171f6e8bec9572e75f192967","observation_id":"e6fd6099-1cf8-45f5-820c-ab3e9b1da9c1","resolution":{"observed_at":"2026-07-02T15:27:06.046624Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2606.06302","last_updated":"2026-06-04T15:41:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-06-04T15:41:27Z","title":"Tangram: Unlocking Non-Uniform KV Cache for Efficient Multi-turn LLM Serving","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:46.234576Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2606.06302"},"observation_digest":"sha256:8aa29ca744d8321bb19ab6b2377a831f6d1fd9d8adb404a05b3f3830701cd398","observation_id":"093eaa09-3de9-45f5-b0a4-dcbc4333307b","resolution":{"observed_at":"2026-07-02T12:06:56.124678Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2606.10493","last_updated":"2026-06-09T07:17:34Z","snapshot_observed_at":"2026-08-02T12:04:13.074478Z","submitted_at":"2026-06-09T07:17:34Z","title":"Achieving Cloud-Grade SLOs for Local Mixture-of-Experts Inference through CPU-GPU Hybrid Design","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-27T12:06:46.806138Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2606.10493"},"observation_digest":"sha256:82a85855ec74becafc0c9cd26d963db5eb4b69831114ef764ade67723f0d3980","observation_id":"b62544c9-918c-4d02-a3a5-d65f9598d3cc","resolution":{"observed_at":"2026-07-03T07:27:44.463957Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2606.22541","last_updated":"2026-06-21T14:57:45Z","snapshot_observed_at":"2026-07-06T23:57:23.574340Z","submitted_at":"2026-06-21T14:57:45Z","title":"ASAP: A Disaggregated and Asynchronous Inference System for MoE Prefill","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-26T09:42:39.573568Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2606.22541"},"observation_digest":"sha256:ec0125bd4751540f5aff64b92082447bd7ba5723919dd26a1ef4d52f6600a6b7","observation_id":"5a4533d2-5f58-4f84-aa5c-5b6e1c69650a","resolution":{"observed_at":"2026-07-04T09:39:46.785034Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2606.22983","last_updated":"2026-06-22T08:04:12Z","snapshot_observed_at":"2026-07-06T23:57:47.047975Z","submitted_at":"2026-06-22T08:04:12Z","title":"LiveServe: Interaction-Aware Serving for Real-Time Omni-Modal LLMs","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-26T07:26:07.356352Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2606.22983"},"observation_digest":"sha256:9cbe885d4e6e9da7e3b84c6e05717cfaa8780d60c5ae2886dbb64b1008b698a4","observation_id":"001e5b19-1864-4616-932c-50058cf2494e","resolution":{"observed_at":"2026-07-04T11:59:50.605397Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2606.26666","last_updated":"2026-07-01T08:54:32Z","snapshot_observed_at":"2026-08-03T23:08:24.179797Z","submitted_at":"2026-06-25T06:56:43Z","title":"PersistentKV: Page-Aware Decode Scheduling for Long-Context LLM Serving on Commodity GPUs","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-07-02T21:04:43.874528Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2606.26666"},"observation_digest":"sha256:c57b051c7aa147eccd4e92191f2178085c3052c4bac9a8802366845783702869","observation_id":"55d8ed2b-9a86-44fa-8e1e-ed2a07997247","resolution":{"observed_at":"2026-07-02T21:07:23.425299Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2606.28565","last_updated":"2026-07-02T09:24:03Z","snapshot_observed_at":"2026-08-03T10:20:30.500902Z","submitted_at":"2026-06-26T19:43:38Z","title":"KernelSight-LM: A Kernel-Level LLM Inference Simulator","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-30T00:48:19.207465Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2606.28565"},"observation_digest":"sha256:b9c9a5c63de336aeb659cc96cd218d60165c41ab3433b9617353788485d4941e","observation_id":"91a132f0-dacf-4838-a14c-0b6a4a24dd4d","resolution":{"observed_at":"2026-07-01T16:15:49.765804Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2606.28565","last_updated":"2026-07-02T09:24:03Z","snapshot_observed_at":"2026-08-03T10:20:30.500902Z","submitted_at":"2026-06-26T19:43:38Z","title":"KernelSight-LM: A Kernel-Level LLM Inference Simulator","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-07-03T23:09:38.092583Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2606.28565"},"observation_digest":"sha256:0adda0f06be5cad52e05998c93d517062c8c3ca25123ba8421e39631f8ed6cbd","observation_id":"67be4816-2fe7-469e-b273-0698ee5d3ed6","resolution":{"observed_at":"2026-07-03T23:19:02.751862Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-07-12T17:31:15.972423Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.02523","last_updated":"2026-05-06T17:03:24Z","snapshot_observed_at":"2026-07-12T17:31:15.741658Z","submitted_at":"2026-05-06T17:03:24Z","title":"Edge-Deployable LLM Fine-Tuning on a Single GPU for Telecom Network Troubleshooting","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-07-12T17:31:15.972423Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2607.02523"},"observation_digest":"sha256:e834ee91bd091afa1ea8fa753844f3f2f019d7b14be58c51de1139f25670a471","observation_id":"db4971cb-75b1-46cc-a5bc-9efe313ef085","resolution":{"observed_at":"2026-07-12T17:31:15.972423Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-07-12T09:50:23.266920Z","title":"S., and Ramjee, R.Sarathi: Efficient llm inference by piggybacking decodes with chunked prefills.arXiv preprint arXiv:2308.16369(2023)","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.02574","last_updated":"2026-06-30T16:12:40Z","snapshot_observed_at":"2026-07-12T09:50:22.497230Z","submitted_at":"2026-06-30T16:12:40Z","title":"From Tensor Buffer to Distributed Memory Hierarchy: A Survey of KV Cache Management for LLM Serving","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-07-12T09:50:23.266920Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2607.02574"},"observation_digest":"sha256:def9f7c36fa4d972b64d64d570a03638c527fc19af144d3d8ee7b063ff97385e","observation_id":"4c91eeed-246f-4938-89dd-1552cefc34ba","resolution":{"observed_at":"2026-07-12T09:50:23.266920Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2607.07862","last_updated":"2026-07-08T18:54:08Z","snapshot_observed_at":"2026-08-02T18:07:48.851195Z","submitted_at":"2026-07-08T18:54:08Z","title":"CTA-Pipelining: A Latency-Oriented Spatial Scaling Method for Multi-GPU Systems","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-07-10T16:30:09.561233Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2607.07862"},"observation_digest":"sha256:5868ffc5863b76af501b698b604239f2bdcc857117c52ab2c9fc8aa32bd8618d","observation_id":"673c8a16-7e9e-488c-813f-c4e0314ff1de","resolution":{"observed_at":"2026-07-10T16:37:23.050625Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-07-13T05:43:12.690359Z","title":"Sarathi: Efficient llm inference by piggybacking decodes with chunked prefills,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.08930","last_updated":"2026-07-09T20:48:08Z","snapshot_observed_at":"2026-08-05T05:08:33.855631Z","submitted_at":"2026-07-09T20:48:08Z","title":"BlockServe: Block-Grained Continuous Batching for High-Throughput Diffusion LLM Serving","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-07-13T05:43:12.690359Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2607.08930"},"observation_digest":"sha256:6c22590549c2799f46745641bf6b5cbac414c1d4c2f7d60154a0e992e9a92de2","observation_id":"d4ad8f6e-ec23-4eec-be11-31e37ec220d3","resolution":{"observed_at":"2026-07-13T05:43:12.690359Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-07-14T07:51:15.549754Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.10987","last_updated":"2026-07-13T01:19:13Z","snapshot_observed_at":"2026-08-02T08:08:42.284530Z","submitted_at":"2026-07-13T01:19:13Z","title":"[AAFLOW+] Stateful Operator Abstraction with Zero-Copy Distributed KV Cache Orchestration for Multi-Agent Workflows","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-07-14T07:51:15.549754Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2607.10987"},"observation_digest":"sha256:447e692294beec8229cb7413c7b9eed360a1d6fd2202404446abd54f810e01da","observation_id":"dfd6711a-345d-4153-b0ce-deb89e82792a","resolution":{"observed_at":"2026-07-14T07:51:15.549754Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-01T21:32:01.677558Z","title":"Gulavani, and Ramachandran Ramjee","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.16074","last_updated":"2026-07-17T15:58:20Z","snapshot_observed_at":"2026-08-04T20:41:56.478013Z","submitted_at":"2026-07-17T15:58:20Z","title":"JoyNexus: Service-Oriented Multi-Tenant Post-Training for VLA Models","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-01T21:32:01.677558Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2607.16074"},"observation_digest":"sha256:40d9e20a82c066d80ec4b841d4d95fda0702f95f06bf0b5303ecfc8479631bab","observation_id":"7ce8cbfb-3ea3-4862-a784-c002cdd0a97c","resolution":{"observed_at":"2026-08-01T21:32:01.677558Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-01T21:12:12.115153Z","title":"Agrawal, A","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.16183","last_updated":"2026-07-17T17:57:49Z","snapshot_observed_at":"2026-08-03T17:36:07.986054Z","submitted_at":"2026-07-17T17:57:49Z","title":"A Blueprint for Equilibrium-Based Differentiable Continuous-Variable Thermodynamic Computing","version":1},"reference_index":104,"source":"pdf_text","source_observed_at":"2026-08-01T21:12:12.115153Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2607.16183"},"observation_digest":"sha256:8eaf12f4bad14a19bc838668ce9947f0cb6e8769369c2b805b536924a3ddf52e","observation_id":"25433c1c-42f0-4738-9c5a-b4d897fd6b82","resolution":{"observed_at":"2026-08-01T21:12:12.115153Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-03T01:35:53.234326Z","title":"Gulavani, and Ramachandran Ramjee","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.28848","last_updated":"2026-07-30T21:20:21Z","snapshot_observed_at":"2026-08-05T17:13:14.392888Z","submitted_at":"2026-07-30T21:20:21Z","title":"DeltaServe: Host-Agnostic Co-Serving of Inference and Fine-Tuning for LLMs","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-03T01:35:53.234326Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2607.28848"},"observation_digest":"sha256:e264f04ade31f6d92c490145ca6b8c4616ab8d9a758b4688703745418d78606f","observation_id":"986a2d1d-7550-43df-a407-ef37ae9c12e2","resolution":{"observed_at":"2026-08-03T01:35:53.234326Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-04T01:44:54.104824Z","title":"arXiv preprint arXiv:2308.16369","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00026","last_updated":"2026-07-11T07:41:28Z","snapshot_observed_at":"2026-08-05T16:14:36.059489Z","submitted_at":"2026-07-11T07:41:28Z","title":"Request-Level Energy Attribution for Batched LLM Serving","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-04T01:44:54.104824Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2608.00026"},"observation_digest":"sha256:aa21c39ab38a0ddf53e8ed5078331d9d408da9ecd9936a09b52112211a014aa0","observation_id":"b83278f4-402a-4a2b-864e-341bc3b30f79","resolution":{"observed_at":"2026-08-04T01:44:54.104824Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-04T00:44:24.913037Z","title":"Agrawal, A","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.00337","last_updated":"2026-07-31T22:58:59Z","snapshot_observed_at":"2026-08-05T16:15:28.991550Z","submitted_at":"2026-07-31T22:58:59Z","title":"Action Chunk Scheduling for Batched Robot Policy Serving","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-04T00:44:24.913037Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2608.00337"},"observation_digest":"sha256:003de0226f7572f6be7ef8c3ab2b8b283ede278e1269a52e683c22a30183aceb","observation_id":"ab8ae17b-813b-4f18-8c37-91a3224871ba","resolution":{"observed_at":"2026-08-04T00:44:24.913037Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-04T18:49:21.474198Z","title":"SARATHI: Efficient LLM inference by piggybacking decodes with chunked prefills,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.01891","last_updated":"2026-08-03T08:32:50Z","snapshot_observed_at":"2026-08-05T16:22:03.367895Z","submitted_at":"2026-08-03T08:32:50Z","title":"Energy-Efficient LLM Serving via Disaggregated Attention--FFN and Flexible Frequency Scaling","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-04T18:49:21.474198Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2608.01891"},"observation_digest":"sha256:4c2dab9d9563508d28b1875863ede019de31537464fe6e84de572622c73b835e","observation_id":"6f6ae754-fee4-4a2b-a942-2cc2210b866c","resolution":{"observed_at":"2026-08-04T18:49:21.474198Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-04T10:54:00.298222Z","title":"Susanne Albers","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.02244","last_updated":"2026-08-03T13:55:20Z","snapshot_observed_at":"2026-08-05T16:24:27.236178Z","submitted_at":"2026-08-03T13:55:20Z","title":"Efficiency and Cost Alignment in Batched LLM Serving via Resource-Fair Scheduling","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-04T10:54:00.298222Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2608.02244"},"observation_digest":"sha256:37af20d211491685030dfef567788fc04c082f8d7fddcc91e3fe41a059031a8f","observation_id":"33809743-9fe6-4fe0-afd9-442265d89329","resolution":{"observed_at":"2026-08-04T10:54:00.298222Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2308.16369/citation-record","integrity":"/paper/2308.16369/integrity","json":"/paper/2308.16369/citation-record.json","paper":"/paper/2308.16369"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"https://aws.amazon.com/ codewhisperer/","venue":null,"work_id":"8ceb9062-1081-4032-823d-82de237e4f51","year":null},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:8abb4a80618130394b691d572cbf32dc559b14a104172153bc18e6bc29254268","observation_id":"a105e427-4160-4241-a2a1-abea7e724f96","resolution":{"observed_at":"2026-05-16T06:31:47.522365Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"https://claude.ai","venue":null,"work_id":"f42d5c73-87c4-4047-8114-d692921e1e62","year":null},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:3b536621f6e8d76df2d07efc2a9c4bf27d733b1babebc9a8d35976e3b71710ca","observation_id":"e902f87a-52a2-477f-b2db-bf0cb17a17d1","resolution":{"observed_at":"2026-05-16T06:31:47.528671Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"https://www.bing.com/chat","venue":null,"work_id":"9b6158d3-54e7-4e29-bd55-9fd55925290c","year":null},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:75f9361632c975d92d99114b01a9199df8c83b06db30b59f5f23b86d135af7d9","observation_id":"f5824472-cfc1-4eaa-a535-a0a4fd72e88c","resolution":{"observed_at":"2026-05-16T06:31:47.531607Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"https://character.ai","venue":null,"work_id":"53e8cd23-a2da-4851-ba9c-d27d179df274","year":null},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:de50396613a850b45d92f3ffb67f1a0bbf164afc04001742b06045aaf88a717a","observation_id":"ef6b5aba-023a-4817-bfab-f635250e345c","resolution":{"observed_at":"2026-05-16T06:31:47.534417Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"https://chat.openai.com","venue":null,"work_id":"1d52047a-4bbb-4d45-8130-15e3ce4a1d05","year":null},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:1bd23e0404c68db5446259ea30430f7d2c22a0c2921ba0cae523a37c2e9baac6","observation_id":"56523f58-fab0-4040-aba0-86d409455c8e","resolution":{"observed_at":"2026-05-16T06:31:47.387231Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"https://github.com/NVIDIA/ FasterTransformer","venue":null,"work_id":"17292c03-eb09-41f9-9060-c35d9131dbee","year":null},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:e986627a6a54fb3ee1cd5e5fda17db2421ca09d5c03e6a63db02f4174600921e","observation_id":"e3dd7e90-3bd7-42c4-aba1-130564e9b81b","resolution":{"observed_at":"2026-05-16T06:31:47.390775Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"https://github.com/features/ copilot","venue":null,"work_id":"591f0346-a78e-4a43-ada0-84fcaecea342","year":null},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:6c2678189d9931f4fa95b1b000b9415269ef5234596e7136954234ed45e51c19","observation_id":"43b77621-ebaf-4d1e-b36f-2f25293ab284","resolution":{"observed_at":"2026-05-16T06:31:47.393917Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"https://bard.google.com","venue":null,"work_id":"b421a2d5-1e4e-476f-9656-fc321142a6b0","year":null},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:15ba9db2f1774a44ba5009cd2a835e604c879efc267c1550a4fc4709f975b6f1","observation_id":"a0d36354-b63a-49a9-9f1d-e2ba054a5726","resolution":{"observed_at":"2026-05-16T06:31:47.396924Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"https://komo.ai/","venue":null,"work_id":"08e0ca35-6cdf-4c0b-98db-b3593822129f","year":null},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:0a35003a0c1e84c9e7542e00533da75909ff9d5713e52c38ff643ef0784d4ea5","observation_id":"8f3d80ce-3263-4912-b024-4ea163211f2d","resolution":{"observed_at":"2026-05-16T06:31:47.400069Z","resolver_source":"raw_fallback","status":"parse_uncertain"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"https://huggingface.co/ decapoda-research/llama-13b-hf","venue":null,"work_id":"23526433-9c50-40fb-9738-9fd2a4934b50","year":null},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:ecd67b707a36cd5cfe35795c9ad29b63b96d40d0db5574c4c071e7fe5056db6b","observation_id":"39d7275e-1c3c-40f4-891d-4f19b6eb04f9","resolution":{"observed_at":"2026-05-16T06:31:47.403590Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"https://docs.nvidia.com/deeplearning/ performance/dl-performance-matrix- multiplication/index.html","venue":null,"work_id":"60b038f3-95ba-4cb3-85a3-5404d9abe4d5","year":null},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:cc0ab43fef7d2d44030c9176fd0d23b679eb177df80a68cd907f674f4c0539c5","observation_id":"a093a949-6477-47fb-a696-b07e3dd97d9e","resolution":{"observed_at":"2026-05-16T06:31:47.407072Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"https://github.com/karpathy/nanoGPT","venue":null,"work_id":"c7c06f3a-1b23-4fce-9cda-023ee5c283fa","year":null},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:3226811f5a92fc05b3412133cdf8896f825ff3cb448f5c48eb831a3d0971e2fe","observation_id":"6bde7ef6-0a05-4405-9d02-3e1d2b17f4d0","resolution":{"observed_at":"2026-05-16T06:31:47.410143Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"https: //developer.nvidia.com/nvidia-triton- inference-server","venue":null,"work_id":"71448a87-6a61-4a20-9a22-fcd14e2659f8","year":null},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:2ebe725ec1d767c61e7152dce330c478286ebb567fb06f944a1aa1505feb4983","observation_id":"232a8aaf-560b-433e-a12c-5b90447c3ab3","resolution":{"observed_at":"2026-05-16T06:31:47.413152Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"https://www.theaidream.com/post/openai-gpt- 3-understanding-the-architecture","venue":null,"work_id":"d67f5439-5032-4c6c-8ad3-3839480203fa","year":null},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:65b15f5c792415abb9517bcf378b78b0caa46693416ac1f1b92b2c211da0b90b","observation_id":"881f4fa7-d0ce-40f2-81cd-3e103c0997b8","resolution":{"observed_at":"2026-05-16T06:31:47.416734Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"https://www.perplexity.ai/","venue":null,"work_id":"c0ca74e2-e7de-44a3-ae2e-c9e324ed70c0","year":null},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:24ee4beaf0cedd98f69ce2d9102d1b0ff3e81c468942b972666214e66b41247f","observation_id":"bf913a27-8cc5-4926-9738-1286a09b353f","resolution":{"observed_at":"2026-05-16T06:31:47.420385Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"https://replit.com/site/ ghostwriter","venue":null,"work_id":"cda2c1c6-6c0e-4955-b67a-cc275bb00ba2","year":null},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:9d20581d1d56c37bf98767925da377a6b4d956ef17c495d4b304ec3f78d61c90","observation_id":"8beb3fb5-dab8-4427-abcc-cd05394948b4","resolution":{"observed_at":"2026-05-16T06:31:47.425685Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"https://huggingface.co/ text-generation-inference","venue":null,"work_id":"c6ec3494-0988-43be-a692-14827ba6df72","year":null},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:1d77b9977dc4953244aa307b45fd6bc933120501a279086d31b395b9783e0292","observation_id":"766a7a7a-b3a2-44bc-abe2-146c818423af","resolution":{"observed_at":"2026-05-16T06:31:47.428812Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"https: //blog.gopenai.com/how-to-speed-up-llms- and-use-100k-context-window-all-tricks-in- one-place-ffd40577b4c","venue":null,"work_id":"19d8bb56-0a8f-4db6-8656-248d579527c9","year":null},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:7435a6206a50bc6f2f54f5519a55506a527288dd853d9fdcaf890a3c70f5b29e","observation_id":"d844abbd-a60c-4928-953c-92d11bd7164b","resolution":{"observed_at":"2026-05-16T06:31:47.432185Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"https: //core.vmware.com/blog/using-nvidias-aiml- frameworks-generative-ai-vmware-vsphere","venue":null,"work_id":"49604da1-c0c9-4a0f-8f21-29cab16d7ab0","year":null},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:baac8367217517004b384cefd575706e6042cefb6d9af940dd6a7996bb854712","observation_id":"70ef3112-31bb-427e-a177-33cf0f40ad31","resolution":{"observed_at":"2026-05-16T06:31:47.435797Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"https://github.com/vllm-project/vllm","venue":null,"work_id":"cd8f0b05-053e-42b3-a28d-f18ce655ffe3","year":null},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:473f412b94d61eae59ef416faccaec1e4ad51d75d90a4270b4a075c8eefc140e","observation_id":"e290db03-dd13-44ba-83d3-f8f0d22b44a2","resolution":{"observed_at":"2026-05-16T06:31:47.439266Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"https://facebookresearch.github.io/xformers/ components/ops.html","venue":null,"work_id":"6d1f289f-f0e1-4cd5-943f-80edcf8f0e40","year":null},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:ff6f534ae82d0181edcd7fa2c04f834e4cb794cd6a373091609dae60d3c289b4","observation_id":"57426900-144d-47fd-b70b-4beaf116d805","resolution":{"observed_at":"2026-05-16T06:31:47.442729Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"https://you.com/","venue":null,"work_id":"d56689d6-2e60-4cf6-9b78-cf42423b28f6","year":null},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:cd851c507990576585e49cd32d01cf2e867997b297c1307ff22fe00b2c69d1e9","observation_id":"b3108b96-2be2-4ef8-a367-7d3d33348065","resolution":{"observed_at":"2026-05-16T06:31:47.445810Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Ef- ficient large scale language modeling with mixtures of experts","venue":null,"work_id":"d43d1ba1-67b6-4fd3-9324-9d9c06f11408","year":2022},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:bb672c795e151706511417b05fb3692e0a6e2d1abfae6e827d88c59d9b054652","observation_id":"46f32074-a200-4aa0-b324-b8e31b72980c","resolution":{"observed_at":"2026-05-16T06:31:47.448872Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Varuna: scal- able, low-cost training of massive deep learning models","venue":null,"work_id":"684af517-a351-40ac-8bd6-64120036be5b","year":2022},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:876102dd0e8d5f534751fc615d53f45810305e16b044f92535b3ccf8d5dd7ac3","observation_id":"369ca656-70c0-469a-98cc-3e165495f53a","resolution":{"observed_at":"2026-05-16T06:31:47.451841Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Language models are few-shot learn- ers","venue":null,"work_id":"601613c9-ef90-4405-bba4-608852cdeb89","year":1901},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:b45b5940c30577a0e427e1aec26449dff56a6deddbbe1ae2651aa7a04e484ef1","observation_id":"c134673a-3b18-404f-bd34-597431574682","resolution":{"observed_at":"2026-05-16T06:31:47.455157Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2204.02311","last_updated":"2022-10-05T06:02:24Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-04-05T16:11:45Z","title":"PaLM: Scaling Language Modeling with Pathways","version":5},"cited_work":{"arxiv_id":"2204.02311","doi":"10.48550/arxiv.2204.02311","metadata_source":"pith","pith_arxiv_id":"2204.02311","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PaLM: Scaling Language Modeling with Pathways","venue":"cs.CL","work_id":"a94f3ef7-2c49-4445-93fe-6ec16aafd966","year":2022},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"cited_paper":"/paper/2204.02311","citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:9545bb61d2afef2399e5cd493edcae9e8c6e16d90d501f34b73b3b9f74d29b22","observation_id":"b517b334-edd3-480e-992c-a8a2bb97ca17","resolution":{"observed_at":"2026-05-16T06:31:47.370660Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-05-21T12:25:21.411055+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-21T12:25:21.411055+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Clipper: 15 A {Low-Latency} online prediction serving system","venue":null,"work_id":"c2539ae8-01b1-41dd-a283-5e8384d3c9c3","year":2017},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:001fc65983d9a5a67b4bbeafde5a7728d68b8ea10dbad73559c5798783aafc06","observation_id":"1b9d845d-20b8-4a40-a068-9ae8477833d3","resolution":{"observed_at":"2026-05-16T06:31:47.458092Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Flashattention-2: Faster attention with better parallelism and work partitioning","venue":null,"work_id":"a116cad0-11b9-45ad-a4dc-a9744e890fff","year":2023},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:2cf65c5f36e1aed2118de2f27b911d7d06a74a00e98115e968b36763effaa320","observation_id":"bffd7178-3a30-4336-84cb-641bfa3001d5","resolution":{"observed_at":"2026-05-16T06:31:47.461983Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Fu, Stefano Ermon, Atri Rudra, and Christopher Ré","venue":null,"work_id":"b9b497aa-64a4-40e4-a175-d826e777acdd","year":2022},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:5cdd44739d7448d8697cb12b281b75145b6d83ceb5685856471bb0c1ee1c1b5f","observation_id":"78cb58de-9125-440b-8f0a-143734629bf6","resolution":{"observed_at":"2026-05-16T06:31:47.465251Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Llm.int8(): 8-bit matrix multiplication for transformers at scale","venue":null,"work_id":"7e60b8bc-a4fb-4385-98c8-f01239ab6e96","year":2022},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:a7c1e8019e3566dfd1116876dc68e3d4889a82f8564d0f676290120291dd648c","observation_id":"49d17720-2838-4863-96b5-484def4aff56","resolution":{"observed_at":"2026-05-16T06:31:47.469018Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Qlora: Efficient finetuning of quan- tized llms","venue":null,"work_id":"674825fd-6768-4267-9396-62c8249e05dc","year":2023},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:198fb10fe788b9824e594aa4836f385de313ad9f9daa423ef783b403f0737159","observation_id":"95821179-3f0a-46cf-99c3-c3627a8a6e56","resolution":{"observed_at":"2026-05-16T06:31:47.472500Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Gptq: Accurate post-training quantization for generative pre-trained transformers","venue":null,"work_id":"3c488d35-6f66-410d-a82c-34d132864808","year":2023},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:6643ceb63215626b6980e9a8c0988bc7f03de2753c3d853abf501922eb0bc810","observation_id":"f7dc33df-c59a-41fb-b718-2479d2455d15","resolution":{"observed_at":"2026-05-16T06:31:47.476366Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Lee, Anjali Sridhar, Shruti Bhosale, Carole-Jean Wu, and Benjamin Lee","venue":null,"work_id":"b72b067f-e9f5-4606-815e-0a4c34ef3abd","year":2023},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:2ce93a611a1b119d21bb8529cc1ac332b1423c1bfa67c5f79f75e3192d14f22c","observation_id":"7c1d80fe-473a-49b0-b9fa-b0bb66b0011c","resolution":{"observed_at":"2026-05-16T06:31:47.479244Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Gpipe: Effi- cient training of giant neural networks using pipeline parallelism","venue":null,"work_id":"5da19d1a-5f16-4003-9d22-36ef93b1c700","year":2019},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:92e6999a9346f190fb43417a769d003632a9ddaeff3c2931013ca8253f5fa6b6","observation_id":"3aaddcb9-7463-4584-b7fa-8b24980c6c5c","resolution":{"observed_at":"2026-05-16T06:31:47.483062Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2001.08361","last_updated":"2020-01-23T03:59:20Z","snapshot_observed_at":"2026-07-06T08:52:12.656082Z","submitted_at":"2020-01-23T03:59:20Z","title":"Scaling Laws for Neural Language Models","version":1},"cited_work":{"arxiv_id":"2001.08361","doi":"10.1145/3616855.3635845","metadata_source":"pith","pith_arxiv_id":"2001.08361","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Scaling Laws for Neural Language Models","venue":"cs.LG","work_id":"b7dd8749-9c45-4977-ab9b-64478dce1ae8","year":2020},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"cited_paper":"/paper/2001.08361","citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:f434d1a90a982322af7d906f379d1c250e7386df374d671fd11b43dc45ff1036","observation_id":"df99e933-060a-4123-968b-bfe415ee2feb","resolution":{"observed_at":"2026-05-16T06:31:47.383653Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Accelerating distributed MoE training and inference with lina","venue":null,"work_id":"5cb8ba20-ee72-4a7d-8a7f-ce0d3de9a20e","year":2023},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:7999ac01106e37cee3aa98b679e4a6400d2c03044b15ac7cfb11ca30e9d05bcc","observation_id":"91b23fbd-e8cd-4d7b-bc13-55e0957c0490","resolution":{"observed_at":"2026-05-16T06:31:47.486920Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Pipedream: gen- eralized pipeline parallelism for dnn training","venue":null,"work_id":"a248631d-1c49-4998-99ad-8d855e8b7959","year":2019},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:41e5d388a63a99a9017a7c374d0589a2f4cb254aa26accfc9d99b14c04402dae","observation_id":"4998e0ee-cb84-48de-ac81-d8c86618bc94","resolution":{"observed_at":"2026-05-16T06:31:47.490548Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":"2303.08774","doi":"10.1002/tea.20265","metadata_source":"pith","pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GPT-4 Technical Report","venue":"cs.CL","work_id":"b928e041-6991-4c08-8c81-0359e4097c7b","year":2023},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:c2dc84dc7d0676822aac027995c9a1870fa97803108f76410d1b8e8834929231","observation_id":"3e7aea7c-03f5-435c-a20b-bb02b85b1657","resolution":{"observed_at":"2026-05-16T06:31:47.363642Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Efficiently scaling transformer inference","venue":null,"work_id":"a0acddf6-3d2f-43fe-8921-66e75091c3b5","year":2022},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:c8e84956fb160ffcbf8b347fe8c70f0aa3beecc688ae87d09280f1ce77146272","observation_id":"5acd7e4e-86ab-4722-8564-226b18a2a283","resolution":{"observed_at":"2026-05-16T06:31:47.493754Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Rabe and Charles Staats","venue":null,"work_id":"fa9198a6-1489-41a1-aaa3-57c940a0f605","year":2022},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:942b1283b39295d777bdd3d35c83b2a22c347939ef3e6f1a2c3a05c211cf005c","observation_id":"be8945db-5e77-470a-89e5-db6a777c31e8","resolution":{"observed_at":"2026-05-16T06:31:47.497115Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Fast transformer decoding: One write- head is all you need","venue":null,"work_id":"e3c08f07-6b24-407f-8923-e64ea802ed57","year":2019},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:1ce113a52d70bfdd900281ba722389df5d7c1b9692077a5bab1a1d147c3c94fc","observation_id":"217f0a45-d75c-46a4-82a5-905704a3d4f9","resolution":{"observed_at":"2026-05-16T06:31:47.500267Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Fu, Zhiqiang Xie, Beidi Chen, Clark Barrett, Joseph E","venue":null,"work_id":"514cda8b-6807-4077-b854-3d1c65d16248","year":2023},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:2637d1c3cc5beeda2c5034baa868e431105e6dbadef4c1dff22b660e89d8d33a","observation_id":"072d2831-7e23-465d-abb9-25b1180f7b16","resolution":{"observed_at":"2026-05-16T06:31:47.504655Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.08053","last_updated":"2020-03-13T23:45:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-09-17T19:42:54Z","title":"Megatron-LM: Training Multi-Billion Parameter Language Models Using Model Parallelism","version":4},"cited_work":{"arxiv_id":"1909.08053","doi":"10.48550/arxiv.1909.08053","metadata_source":"pith","pith_arxiv_id":"1909.08053","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Megatron-LM: Training Multi-Billion Parameter Language Models Using Model Parallelism","venue":"cs.CL","work_id":"c888e6d1-0b1d-43d6-9ef5-f0912a0efa1b","year":2019},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"cited_paper":"/paper/1909.08053","citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:f5f9a18febfb89fa24da675c4433bfce36a7dea7307ffffb36a310ec74867697","observation_id":"a8d9d635-39ef-4241-8838-2d62a9677aad","resolution":{"observed_at":"2026-05-16T06:31:47.377211Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-07-09T10:48:33.392193+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-09T10:48:33.392193+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Retentive network: A successor to transformer for large language models","venue":null,"work_id":"55f5b239-668c-4843-91fd-cbb279e249a0","year":2023},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:88cca76c25ffb541f9ab665126adf470fd7dbc8e1076f503166f099ba5071515","observation_id":"30a76471-d0e8-45a1-871f-a6cae359a10e","resolution":{"observed_at":"2026-05-16T06:31:47.509432Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Chi, Tat- sunori Hashimoto, Oriol Vinyals, Percy Liang, Jeff Dean, and William Fedus","venue":null,"work_id":"a11a96fe-ae3a-45e2-9569-f24e96d8b753","year":2022},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:4b7c4ed78491b6244c184d1295c8c085ee7e8912f012cda31ce58006fa3ee5cb","observation_id":"6d805c9c-26f5-4ccb-9f51-85326a5dead2","resolution":{"observed_at":"2026-05-16T06:31:47.512757Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Fast distributed inference serving for large language models","venue":null,"work_id":"f970c665-d887-4811-897f-ecfdc5799c00","year":2023},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:b00e16e080aea65e79d8c84a3b7643392ce3f112e6c66cdccf3e2326185377bc","observation_id":"9f7c027a-6cd8-4c75-b390-5e21be4f88f1","resolution":{"observed_at":"2026-05-16T06:31:47.515978Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Smoothquant: Accu- rate and efficient post-training quantization for large language models","venue":null,"work_id":"e29a145a-b615-40f6-8bc3-d3df482bc985","year":2023},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:e841ca2d8b7b5b1ff008f095ebb07a79a2ebef159c577deecba1464395b178a0","observation_id":"88eb4b57-e784-4861-8c1a-1ad18faf17fe","resolution":{"observed_at":"2026-05-16T06:31:47.519160Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Orca: A distributed serving system for Transformer-Based generative mod- els","venue":null,"work_id":"6346f909-07f4-4a5a-900e-f780d4ea16f5","year":2022},"citing_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-16T06:31:47.330226Z"},"links":{"citing_paper":"/paper/2308.16369"},"observation_digest":"sha256:5bd28f132df46f89134ebb9ced26765bb259d7443a68f8124ae2b9dc9d960f4e","observation_id":"4ea16e24-5c31-490d-9cd4-83eb87925b95","resolution":{"observed_at":"2026-05-16T06:31:47.525359Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills"},"reference_resolution":{"displayed":48,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":1,"unresolved":0,"verified_exact":3,"verified_fuzzy":43},"total_outbound_references":48},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"thesis":"As of 5 August 2026, this Paper Citation Record lists 48 of 48 outbound references and 61 inbound Pith citation observations for arXiv:2308.16369."}