{"as_of":"2026-08-05T14:20:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:36f57caa4dec7568e982d46b540f9fd3f20a0189ed2542a4e82267a068b6abba","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":31,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":31,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-05T06:32:48.257954+00:00","state":"measured"},{"denominator":31,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":31,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T20:53:30.772845Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"pith","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":15,"observed_at":"2026-08-05T02:28:24.338817Z","source":"pith"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2403.02310","last_updated":"2024-06-17T21:10:46Z","snapshot_observed_at":"2026-07-06T17:39:24.823145Z","submitted_at":"2024-03-04T18:47:08Z","title":"Taming Throughput-Latency Tradeoff in LLM Inference with Sarathi-Serve","version":3},"cited_work":{"arxiv_id":"2403.02310","doi":"10.48550/arxiv.2403.02310","metadata_source":"pith","pith_arxiv_id":"2403.02310","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gulavani, Alexey Tumanov, and Ramachandran Ramjee","venue":"cs.LG","work_id":"07bbeba4-bd20-4699-9af8-2d24435f090a","year":2024},"citing_paper":{"arxiv_id":"2404.14294","last_updated":"2024-07-19T04:47:36Z","snapshot_observed_at":"2026-07-06T18:03:47.096406Z","submitted_at":"2024-04-22T15:53:08Z","title":"A Survey on Efficient Inference for Large Language Models","version":3},"reference_index":283,"source":"pdf_text","source_observed_at":"2026-05-15T02:39:33.007894Z"},"links":{"cited_paper":"/paper/2403.02310","citing_paper":"/paper/2404.14294"},"observation_digest":"sha256:d4d70e2392cde696ba5351a4252910e99a1c24f82d65a3397fa2b58c2bc59416","observation_id":"8f3425e2-02d2-4560-9573-4846b10184c5","resolution":{"observed_at":"2026-05-15T02:39:33.509031Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.02310","last_updated":"2024-06-17T21:10:46Z","snapshot_observed_at":"2026-07-06T17:39:24.823145Z","submitted_at":"2024-03-04T18:47:08Z","title":"Taming Throughput-Latency Tradeoff in LLM Inference with Sarathi-Serve","version":3},"cited_work":{"arxiv_id":"2403.02310","doi":"10.48550/arxiv.2403.02310","metadata_source":"pith","pith_arxiv_id":"2403.02310","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gulavani, Alexey Tumanov, and Ramachandran Ramjee","venue":"cs.LG","work_id":"07bbeba4-bd20-4699-9af8-2d24435f090a","year":2024},"citing_paper":{"arxiv_id":"2505.09999","last_updated":"2026-05-11T03:41:00Z","snapshot_observed_at":"2026-07-30T05:05:51.180006Z","submitted_at":"2025-05-15T06:24:08Z","title":"ServeGen: Workload Characterization and Generation of Large Language Model Serving in Production","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-22T15:42:05.266854Z"},"links":{"cited_paper":"/paper/2403.02310","citing_paper":"/paper/2505.09999"},"observation_digest":"sha256:895abb6d566c593308036de65f4a8e385912a7e352fe9477f96e9566f1e85489","observation_id":"1e99bc29-4825-41dc-bf6e-482fac912f0b","resolution":{"observed_at":"2026-05-22T15:44:58.114111Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.02310","last_updated":"2024-06-17T21:10:46Z","snapshot_observed_at":"2026-07-06T17:39:24.823145Z","submitted_at":"2024-03-04T18:47:08Z","title":"Taming Throughput-Latency Tradeoff in LLM Inference with Sarathi-Serve","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.02310","snapshot_observed_at":"2026-08-04T20:53:30.772845Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.08309","last_updated":"2025-09-10T06:06:51Z","snapshot_observed_at":"2026-08-04T20:53:29.155948Z","submitted_at":"2025-09-10T06:06:51Z","title":"Hetis: Serving LLMs in Heterogeneous GPU Clusters with Fine-grained and Dynamic Parallelism","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-04T20:53:30.772845Z"},"links":{"cited_paper":"/paper/2403.02310","citing_paper":"/paper/2509.08309"},"observation_digest":"sha256:22f2015c208946bd10a852bc5b94d16bc2f05a79972e7757017177226a652264","observation_id":"b80e34f7-c743-48f1-a56a-f98d807df92c","resolution":{"observed_at":"2026-08-04T20:53:30.772845Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.02310","last_updated":"2024-06-17T21:10:46Z","snapshot_observed_at":"2026-07-06T17:39:24.823145Z","submitted_at":"2024-03-04T18:47:08Z","title":"Taming Throughput-Latency Tradeoff in LLM Inference with Sarathi-Serve","version":3},"cited_work":{"arxiv_id":"2403.02310","doi":"10.48550/arxiv.2403.02310","metadata_source":"pith","pith_arxiv_id":"2403.02310","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gulavani, Alexey Tumanov, and Ramachandran Ramjee","venue":"cs.LG","work_id":"07bbeba4-bd20-4699-9af8-2d24435f090a","year":2024},"citing_paper":{"arxiv_id":"2512.09427","last_updated":"2026-04-21T07:27:04Z","snapshot_observed_at":"2026-07-06T22:38:35.387503Z","submitted_at":"2025-12-10T08:52:20Z","title":"ODMA: On-Demand Memory Allocation Strategy for LLM Serving on LPDDR-Class Accelerators","version":5},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-16T23:54:08.052482Z"},"links":{"cited_paper":"/paper/2403.02310","citing_paper":"/paper/2512.09427"},"observation_digest":"sha256:ffb055bf7f7e7cd712e4bb0cfde73afa4f97e85df40bb38b31fa62f78490894e","observation_id":"1a28d3b4-f021-4bb3-b926-0ec0c9ee9ddc","resolution":{"observed_at":"2026-05-16T23:58:42.965092Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.02310","last_updated":"2024-06-17T21:10:46Z","snapshot_observed_at":"2026-07-06T17:39:24.823145Z","submitted_at":"2024-03-04T18:47:08Z","title":"Taming Throughput-Latency Tradeoff in LLM Inference with Sarathi-Serve","version":3},"cited_work":{"arxiv_id":"2403.02310","doi":"10.48550/arxiv.2403.02310","metadata_source":"pith","pith_arxiv_id":"2403.02310","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gulavani, Alexey Tumanov, and Ramachandran Ramjee","venue":"cs.LG","work_id":"07bbeba4-bd20-4699-9af8-2d24435f090a","year":2024},"citing_paper":{"arxiv_id":"2601.14910","last_updated":"2026-04-28T03:10:59Z","snapshot_observed_at":"2026-08-01T18:54:26.265127Z","submitted_at":"2026-01-21T11:47:56Z","title":"PipeWeave: Synergizing Analytical and Learning Models for Unified GPU Performance Prediction","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-16T12:45:27.028757Z"},"links":{"cited_paper":"/paper/2403.02310","citing_paper":"/paper/2601.14910"},"observation_digest":"sha256:822980c19aa2ae570a513197f7fb0da9be366757f047e18626c1594ba643f1a6","observation_id":"97bcbd00-63ac-4df6-83be-2e57677b482c","resolution":{"observed_at":"2026-05-16T12:47:54.091286Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.02310","last_updated":"2024-06-17T21:10:46Z","snapshot_observed_at":"2026-07-06T17:39:24.823145Z","submitted_at":"2024-03-04T18:47:08Z","title":"Taming Throughput-Latency Tradeoff in LLM Inference with Sarathi-Serve","version":3},"cited_work":{"arxiv_id":"2403.02310","doi":"10.48550/arxiv.2403.02310","metadata_source":"pith","pith_arxiv_id":"2403.02310","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gulavani, Alexey Tumanov, and Ramachandran Ramjee","venue":"cs.LG","work_id":"07bbeba4-bd20-4699-9af8-2d24435f090a","year":2024},"citing_paper":{"arxiv_id":"2604.17353","last_updated":"2026-04-19T09:59:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-19T09:59:35Z","title":"Hive: A Multi-Agent Infrastructure for Algorithm- and Task-Level Scaling","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T06:31:36.776819Z"},"links":{"cited_paper":"/paper/2403.02310","citing_paper":"/paper/2604.17353"},"observation_digest":"sha256:9263a89fb94c91713d1b78b72df59ecb0abdb9c6b808d79360b27f3323945505","observation_id":"c0e65acb-5eca-404e-823f-06ae489253f3","resolution":{"observed_at":"2026-05-10T06:36:36.672643Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.02310","last_updated":"2024-06-17T21:10:46Z","snapshot_observed_at":"2026-07-06T17:39:24.823145Z","submitted_at":"2024-03-04T18:47:08Z","title":"Taming Throughput-Latency Tradeoff in LLM Inference with Sarathi-Serve","version":3},"cited_work":{"arxiv_id":"2403.02310","doi":"10.48550/arxiv.2403.02310","metadata_source":"pith","pith_arxiv_id":"2403.02310","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gulavani, Alexey Tumanov, and Ramachandran Ramjee","venue":"cs.LG","work_id":"07bbeba4-bd20-4699-9af8-2d24435f090a","year":2024},"citing_paper":{"arxiv_id":"2604.24203","last_updated":"2026-04-27T09:07:15Z","snapshot_observed_at":"2026-07-06T23:10:16.426836Z","submitted_at":"2026-04-27T09:07:15Z","title":"Agentic Witnessing: Pragmatic and Scalable TEE-Enabled Privacy-Preserving Auditing","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-08T02:57:11.715370Z"},"links":{"cited_paper":"/paper/2403.02310","citing_paper":"/paper/2604.24203"},"observation_digest":"sha256:a0aac6ed7a79abf083c3b82b6ee748a9a06808b8d55695884a385d75303df7d4","observation_id":"9aa1a6b3-df44-4281-9401-d6426f28a503","resolution":{"observed_at":"2026-05-09T00:34:30.279799Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.02310","last_updated":"2024-06-17T21:10:46Z","snapshot_observed_at":"2026-07-06T17:39:24.823145Z","submitted_at":"2024-03-04T18:47:08Z","title":"Taming Throughput-Latency Tradeoff in LLM Inference with Sarathi-Serve","version":3},"cited_work":{"arxiv_id":"2403.02310","doi":"10.48550/arxiv.2403.02310","metadata_source":"pith","pith_arxiv_id":"2403.02310","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gulavani, Alexey Tumanov, and Ramachandran Ramjee","venue":"cs.LG","work_id":"07bbeba4-bd20-4699-9af8-2d24435f090a","year":2024},"citing_paper":{"arxiv_id":"2605.01214","last_updated":"2026-05-02T03:06:02Z","snapshot_observed_at":"2026-08-04T03:54:47.150761Z","submitted_at":"2026-05-02T03:06:02Z","title":"Agentic AI Systems Should Be Designed as Marginal Token Allocators","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-09T15:11:41.570592Z"},"links":{"cited_paper":"/paper/2403.02310","citing_paper":"/paper/2605.01214"},"observation_digest":"sha256:e8ef6c6a6841bd0fc2e69692998f58e38f16f30895a5970ffeb7afda57a5ec54","observation_id":"43579636-675c-44d7-b5d5-786a088a8e5c","resolution":{"observed_at":"2026-05-11T16:46:05.704482Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.02310","last_updated":"2024-06-17T21:10:46Z","snapshot_observed_at":"2026-07-06T17:39:24.823145Z","submitted_at":"2024-03-04T18:47:08Z","title":"Taming Throughput-Latency Tradeoff in LLM Inference with Sarathi-Serve","version":3},"cited_work":{"arxiv_id":"2403.02310","doi":"10.48550/arxiv.2403.02310","metadata_source":"pith","pith_arxiv_id":"2403.02310","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gulavani, Alexey Tumanov, and Ramachandran Ramjee","venue":"cs.LG","work_id":"07bbeba4-bd20-4699-9af8-2d24435f090a","year":2024},"citing_paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-11T02:42:32.739352Z"},"links":{"cited_paper":"/paper/2403.02310","citing_paper":"/paper/2605.07985"},"observation_digest":"sha256:3d1c3b46cbd7f2e3bc674d57d4a17a57a4893b4a7997cdce3fa530e91640b68f","observation_id":"01a9a38f-abc6-4197-a51c-c117f799eff8","resolution":{"observed_at":"2026-05-11T02:45:58.613756Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.02310","last_updated":"2024-06-17T21:10:46Z","snapshot_observed_at":"2026-07-06T17:39:24.823145Z","submitted_at":"2024-03-04T18:47:08Z","title":"Taming Throughput-Latency Tradeoff in LLM Inference with Sarathi-Serve","version":3},"cited_work":{"arxiv_id":"2403.02310","doi":"10.48550/arxiv.2403.02310","metadata_source":"pith","pith_arxiv_id":"2403.02310","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gulavani, Alexey Tumanov, and Ramachandran Ramjee","venue":"cs.LG","work_id":"07bbeba4-bd20-4699-9af8-2d24435f090a","year":2024},"citing_paper":{"arxiv_id":"2605.07985","last_updated":"2026-05-21T13:49:15Z","snapshot_observed_at":"2026-08-02T04:20:16.015867Z","submitted_at":"2026-05-08T16:44:47Z","title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-22T10:20:45.375375Z"},"links":{"cited_paper":"/paper/2403.02310","citing_paper":"/paper/2605.07985"},"observation_digest":"sha256:8358ab054283ea67b4fb5c988919aff0291424cecb55240052038bbae34b2e6e","observation_id":"74d58261-838b-4fd0-808d-b7d5c2fa05f4","resolution":{"observed_at":"2026-05-22T10:21:23.908790Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.02310","last_updated":"2024-06-17T21:10:46Z","snapshot_observed_at":"2026-07-06T17:39:24.823145Z","submitted_at":"2024-03-04T18:47:08Z","title":"Taming Throughput-Latency Tradeoff in LLM Inference with Sarathi-Serve","version":3},"cited_work":{"arxiv_id":"2403.02310","doi":"10.48550/arxiv.2403.02310","metadata_source":"pith","pith_arxiv_id":"2403.02310","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gulavani, Alexey Tumanov, and Ramachandran Ramjee","venue":"cs.LG","work_id":"07bbeba4-bd20-4699-9af8-2d24435f090a","year":2024},"citing_paper":{"arxiv_id":"2605.17410","last_updated":"2026-05-17T12:11:34Z","snapshot_observed_at":"2026-07-06T23:28:26.397778Z","submitted_at":"2026-05-17T12:11:34Z","title":"Computational Challenges in Token Economics: Bridging Economic Theory and AI System Design","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-20T13:00:56.755353Z"},"links":{"cited_paper":"/paper/2403.02310","citing_paper":"/paper/2605.17410"},"observation_digest":"sha256:da861452f0e447748c5f8c3aa4bba479086ecc42320782520fe528a19e961367","observation_id":"dab488d6-f358-4543-9ec7-46d02a4c250b","resolution":{"observed_at":"2026-05-20T13:03:17.881842Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.02310","last_updated":"2024-06-17T21:10:46Z","snapshot_observed_at":"2026-07-06T17:39:24.823145Z","submitted_at":"2024-03-04T18:47:08Z","title":"Taming Throughput-Latency Tradeoff in LLM Inference with Sarathi-Serve","version":3},"cited_work":{"arxiv_id":"2403.02310","doi":"10.48550/arxiv.2403.02310","metadata_source":"pith","pith_arxiv_id":"2403.02310","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gulavani, Alexey Tumanov, and Ramachandran Ramjee","venue":"cs.LG","work_id":"07bbeba4-bd20-4699-9af8-2d24435f090a","year":2024},"citing_paper":{"arxiv_id":"2605.22733","last_updated":"2026-05-21T17:03:44Z","snapshot_observed_at":"2026-07-06T23:33:04.856017Z","submitted_at":"2026-05-21T17:03:44Z","title":"HarnessAPI: A Skill-First Framework for Unified Streaming APIs and MCP Tools","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-22T05:21:46.992441Z"},"links":{"cited_paper":"/paper/2403.02310","citing_paper":"/paper/2605.22733"},"observation_digest":"sha256:53ebe7365c4c55e565677416937252f6844faeb48cd70ffbc26de84dd2c6bc64","observation_id":"9fce4adf-ae6f-4a6d-a3a4-4ef3950b03fc","resolution":{"observed_at":"2026-05-22T05:24:38.270384Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.02310","last_updated":"2024-06-17T21:10:46Z","snapshot_observed_at":"2026-07-06T17:39:24.823145Z","submitted_at":"2024-03-04T18:47:08Z","title":"Taming Throughput-Latency Tradeoff in LLM Inference with Sarathi-Serve","version":3},"cited_work":{"arxiv_id":"2403.02310","doi":"10.48550/arxiv.2403.02310","metadata_source":"pith","pith_arxiv_id":"2403.02310","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gulavani, Alexey Tumanov, and Ramachandran Ramjee","venue":"cs.LG","work_id":"07bbeba4-bd20-4699-9af8-2d24435f090a","year":2024},"citing_paper":{"arxiv_id":"2605.25550","last_updated":"2026-05-25T08:07:47Z","snapshot_observed_at":"2026-07-31T12:35:46.854130Z","submitted_at":"2026-05-25T08:07:47Z","title":"DisagFusion: Asynchronous Pipeline Parallelism and Elastic Scheduling for Disaggregated Diffusion Serving","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-29T20:46:51.896596Z"},"links":{"cited_paper":"/paper/2403.02310","citing_paper":"/paper/2605.25550"},"observation_digest":"sha256:70eb5dc83dedb25a85cdef1db64298f7f97efa424bc754b9b0a8eff4d2b66381","observation_id":"bd969b09-0574-4b31-9ffa-01e9be155afa","resolution":{"observed_at":"2026-06-29T20:53:57.624787Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.02310","last_updated":"2024-06-17T21:10:46Z","snapshot_observed_at":"2026-07-06T17:39:24.823145Z","submitted_at":"2024-03-04T18:47:08Z","title":"Taming Throughput-Latency Tradeoff in LLM Inference with Sarathi-Serve","version":3},"cited_work":{"arxiv_id":"2403.02310","doi":"10.48550/arxiv.2403.02310","metadata_source":"pith","pith_arxiv_id":"2403.02310","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gulavani, Alexey Tumanov, and Ramachandran Ramjee","venue":"cs.LG","work_id":"07bbeba4-bd20-4699-9af8-2d24435f090a","year":2024},"citing_paper":{"arxiv_id":"2605.27763","last_updated":"2026-05-26T23:22:55Z","snapshot_observed_at":"2026-07-31T18:23:26.053584Z","submitted_at":"2026-05-26T23:22:55Z","title":"A Paired Testing Protocol for Batch-Conditioned Refusal Robustness in LLM Serving","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-06-29T18:03:19.798168Z"},"links":{"cited_paper":"/paper/2403.02310","citing_paper":"/paper/2605.27763"},"observation_digest":"sha256:446c44f7da0e146466983d9e766a13fe8abad530c696f1401758112bfc307498","observation_id":"3748cd8c-fe39-488d-b958-23b9b146be6f","resolution":{"observed_at":"2026-06-29T18:03:47.840958Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.02310","last_updated":"2024-06-17T21:10:46Z","snapshot_observed_at":"2026-07-06T17:39:24.823145Z","submitted_at":"2024-03-04T18:47:08Z","title":"Taming Throughput-Latency Tradeoff in LLM Inference with Sarathi-Serve","version":3},"cited_work":{"arxiv_id":"2403.02310","doi":"10.48550/arxiv.2403.02310","metadata_source":"pith","pith_arxiv_id":"2403.02310","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gulavani, Alexey Tumanov, and Ramachandran Ramjee","venue":"cs.LG","work_id":"07bbeba4-bd20-4699-9af8-2d24435f090a","year":2024},"citing_paper":{"arxiv_id":"2606.00735","last_updated":"2026-05-30T13:57:09Z","snapshot_observed_at":"2026-07-06T23:41:24.911421Z","submitted_at":"2026-05-30T13:57:09Z","title":"ViBE: Co-Optimizing Workload Skew and Hardware Variability for MoE Serving","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-28T18:07:52.038640Z"},"links":{"cited_paper":"/paper/2403.02310","citing_paper":"/paper/2606.00735"},"observation_digest":"sha256:c4dccf63e1011797ef69f7eb8dba10f915714cb1999fced0040d7aaa86a39ce6","observation_id":"ea36a92f-e078-4765-92ad-07ed17376370","resolution":{"observed_at":"2026-07-01T20:46:13.223881Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.02310","last_updated":"2024-06-17T21:10:46Z","snapshot_observed_at":"2026-07-06T17:39:24.823145Z","submitted_at":"2024-03-04T18:47:08Z","title":"Taming Throughput-Latency Tradeoff in LLM Inference with Sarathi-Serve","version":3},"cited_work":{"arxiv_id":"2403.02310","doi":"10.48550/arxiv.2403.02310","metadata_source":"pith","pith_arxiv_id":"2403.02310","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gulavani, Alexey Tumanov, and Ramachandran Ramjee","venue":"cs.LG","work_id":"07bbeba4-bd20-4699-9af8-2d24435f090a","year":2024},"citing_paper":{"arxiv_id":"2606.06453","last_updated":"2026-06-04T17:48:17Z","snapshot_observed_at":"2026-07-06T23:46:15.936608Z","submitted_at":"2026-06-04T17:48:17Z","title":"Vortex: Efficient and Programmable Sparse Attention Serving for AI Agents","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-28T01:07:14.691347Z"},"links":{"cited_paper":"/paper/2403.02310","citing_paper":"/paper/2606.06453"},"observation_digest":"sha256:1b9acaf90d890bbfb38ff17f05bc1d1e958d87d8ce4630d859a5115b45fefe6a","observation_id":"f3070805-629f-4d65-ac3a-4b88cb3866ef","resolution":{"observed_at":"2026-07-02T13:36:59.461051Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.02310","last_updated":"2024-06-17T21:10:46Z","snapshot_observed_at":"2026-07-06T17:39:24.823145Z","submitted_at":"2024-03-04T18:47:08Z","title":"Taming Throughput-Latency Tradeoff in LLM Inference with Sarathi-Serve","version":3},"cited_work":{"arxiv_id":"2403.02310","doi":"10.48550/arxiv.2403.02310","metadata_source":"pith","pith_arxiv_id":"2403.02310","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gulavani, Alexey Tumanov, and Ramachandran Ramjee","venue":"cs.LG","work_id":"07bbeba4-bd20-4699-9af8-2d24435f090a","year":2024},"citing_paper":{"arxiv_id":"2606.18431","last_updated":"2026-06-16T19:25:37Z","snapshot_observed_at":"2026-08-04T00:16:31.977200Z","submitted_at":"2026-06-16T19:25:37Z","title":"Beyond Prediction: Tail-Aware Scheduling for LLM Inference","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-06-27T01:01:34.655458Z"},"links":{"cited_paper":"/paper/2403.02310","citing_paper":"/paper/2606.18431"},"observation_digest":"sha256:4ee2b9619d8bd2ddd0f3a425932c8313fe7a3fab923d31f61c58552d03b5d38f","observation_id":"f11ace76-a222-43e0-a171-c74632f0b438","resolution":{"observed_at":"2026-07-03T20:58:57.825283Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.02310","last_updated":"2024-06-17T21:10:46Z","snapshot_observed_at":"2026-07-06T17:39:24.823145Z","submitted_at":"2024-03-04T18:47:08Z","title":"Taming Throughput-Latency Tradeoff in LLM Inference with Sarathi-Serve","version":3},"cited_work":{"arxiv_id":"2403.02310","doi":"10.48550/arxiv.2403.02310","metadata_source":"pith","pith_arxiv_id":"2403.02310","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gulavani, Alexey Tumanov, and Ramachandran Ramjee","venue":"cs.LG","work_id":"07bbeba4-bd20-4699-9af8-2d24435f090a","year":2024},"citing_paper":{"arxiv_id":"2606.20295","last_updated":"2026-07-24T07:54:17Z","snapshot_observed_at":"2026-08-02T10:48:55.986116Z","submitted_at":"2026-06-18T14:33:09Z","title":"Token-Operations-Oriented Inference Optimization Techniques for Large Models","version":1},"reference_index":183,"source":"pdf_text","source_observed_at":"2026-06-26T16:15:22.543601Z"},"links":{"cited_paper":"/paper/2403.02310","citing_paper":"/paper/2606.20295"},"observation_digest":"sha256:f6c52bdefe63923718a69107974bece200a79673737786e14554b0628883118e","observation_id":"2ca9d91d-efa0-46a4-826a-af33269b5ace","resolution":{"observed_at":"2026-07-04T05:09:36.881011Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.02310","last_updated":"2024-06-17T21:10:46Z","snapshot_observed_at":"2026-07-06T17:39:24.823145Z","submitted_at":"2024-03-04T18:47:08Z","title":"Taming Throughput-Latency Tradeoff in LLM Inference with Sarathi-Serve","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.02310","snapshot_observed_at":"2026-08-02T10:49:20.252837Z","title":"Taming Throughput-Latency Tradeoff in LLM Inference with Sarathi-Serve","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.20295","last_updated":"2026-07-24T07:54:17Z","snapshot_observed_at":"2026-08-02T10:48:55.986116Z","submitted_at":"2026-06-18T14:33:09Z","title":"Token-Operations-Oriented Inference Optimization Techniques for Large Models","version":2},"reference_index":183,"source":"pdf_text","source_observed_at":"2026-08-02T10:49:20.252837Z"},"links":{"cited_paper":"/paper/2403.02310","citing_paper":"/paper/2606.20295"},"observation_digest":"sha256:48890a710c0df07b683cbc41d9491c7abbfa52782b42e7f14ef2021a6d4092b5","observation_id":"67cb0682-5fcb-49f9-9ace-d0fb8fe0be3c","resolution":{"observed_at":"2026-08-02T10:49:20.252837Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.02310","last_updated":"2024-06-17T21:10:46Z","snapshot_observed_at":"2026-07-06T17:39:24.823145Z","submitted_at":"2024-03-04T18:47:08Z","title":"Taming Throughput-Latency Tradeoff in LLM Inference with Sarathi-Serve","version":3},"cited_work":{"arxiv_id":"2403.02310","doi":"10.48550/arxiv.2403.02310","metadata_source":"pith","pith_arxiv_id":"2403.02310","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gulavani, Alexey Tumanov, and Ramachandran Ramjee","venue":"cs.LG","work_id":"07bbeba4-bd20-4699-9af8-2d24435f090a","year":2024},"citing_paper":{"arxiv_id":"2606.22013","last_updated":"2026-06-20T12:34:27Z","snapshot_observed_at":"2026-07-06T23:56:54.959593Z","submitted_at":"2026-06-20T12:34:27Z","title":"Load Testing for Machine Learning Model Serving Systems at Scale","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-26T11:54:48.056285Z"},"links":{"cited_paper":"/paper/2403.02310","citing_paper":"/paper/2606.22013"},"observation_digest":"sha256:8bd753bb52d082de12da21b641779f03184aba248f526a40a14993061b6d4629","observation_id":"bd03649a-e348-4420-b995-3264601b0d54","resolution":{"observed_at":"2026-07-04T08:19:44.171292Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.02310","last_updated":"2024-06-17T21:10:46Z","snapshot_observed_at":"2026-07-06T17:39:24.823145Z","submitted_at":"2024-03-04T18:47:08Z","title":"Taming Throughput-Latency Tradeoff in LLM Inference with Sarathi-Serve","version":3},"cited_work":{"arxiv_id":"2403.02310","doi":"10.48550/arxiv.2403.02310","metadata_source":"pith","pith_arxiv_id":"2403.02310","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gulavani, Alexey Tumanov, and Ramachandran Ramjee","venue":"cs.LG","work_id":"07bbeba4-bd20-4699-9af8-2d24435f090a","year":2024},"citing_paper":{"arxiv_id":"2606.25097","last_updated":"2026-06-23T19:06:42Z","snapshot_observed_at":"2026-08-02T20:00:46.182561Z","submitted_at":"2026-06-23T19:06:42Z","title":"Speculative Decoding at Temperature Zero: A Scoped Safety-Invariance Screen with a 48,072-Sample Expansion","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-26T00:05:06.647295Z"},"links":{"cited_paper":"/paper/2403.02310","citing_paper":"/paper/2606.25097"},"observation_digest":"sha256:aa1568b5fce276a5864c78fb8a3c3ce50b678346d07a4cac0c5f4bbea1d7cb78","observation_id":"c478561c-37b0-40dd-88e9-9e9473efcb13","resolution":{"observed_at":"2026-07-04T16:59:58.321693Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.02310","last_updated":"2024-06-17T21:10:46Z","snapshot_observed_at":"2026-07-06T17:39:24.823145Z","submitted_at":"2024-03-04T18:47:08Z","title":"Taming Throughput-Latency Tradeoff in LLM Inference with Sarathi-Serve","version":3},"cited_work":{"arxiv_id":"2403.02310","doi":"10.48550/arxiv.2403.02310","metadata_source":"pith","pith_arxiv_id":"2403.02310","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gulavani, Alexey Tumanov, and Ramachandran Ramjee","venue":"cs.LG","work_id":"07bbeba4-bd20-4699-9af8-2d24435f090a","year":2024},"citing_paper":{"arxiv_id":"2606.26666","last_updated":"2026-07-01T08:54:32Z","snapshot_observed_at":"2026-08-03T23:08:24.179797Z","submitted_at":"2026-06-25T06:56:43Z","title":"PersistentKV: Page-Aware Decode Scheduling for Long-Context LLM Serving on Commodity GPUs","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-07-02T21:04:43.874528Z"},"links":{"cited_paper":"/paper/2403.02310","citing_paper":"/paper/2606.26666"},"observation_digest":"sha256:6c5ba80e418c1da159c0f9b65d193c6cdf1f6e933ea465a5c62fdf027770494a","observation_id":"0970c22c-a640-4754-a322-26d9829e9c6c","resolution":{"observed_at":"2026-07-02T21:07:23.420190Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.02310","last_updated":"2024-06-17T21:10:46Z","snapshot_observed_at":"2026-07-06T17:39:24.823145Z","submitted_at":"2024-03-04T18:47:08Z","title":"Taming Throughput-Latency Tradeoff in LLM Inference with Sarathi-Serve","version":3},"cited_work":{"arxiv_id":"2403.02310","doi":"10.48550/arxiv.2403.02310","metadata_source":"pith","pith_arxiv_id":"2403.02310","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gulavani, Alexey Tumanov, and Ramachandran Ramjee","venue":"cs.LG","work_id":"07bbeba4-bd20-4699-9af8-2d24435f090a","year":2024},"citing_paper":{"arxiv_id":"2606.28565","last_updated":"2026-07-02T09:24:03Z","snapshot_observed_at":"2026-08-03T10:20:30.500902Z","submitted_at":"2026-06-26T19:43:38Z","title":"KernelSight-LM: A Kernel-Level LLM Inference Simulator","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-30T00:48:19.207465Z"},"links":{"cited_paper":"/paper/2403.02310","citing_paper":"/paper/2606.28565"},"observation_digest":"sha256:906f173ba8bf13b36697ecacf2dce11b1165097e5cd96396a5a6307d68d51a85","observation_id":"cb35a35e-61eb-43a1-8743-61017d6e8f5e","resolution":{"observed_at":"2026-06-30T00:54:06.298161Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.02310","last_updated":"2024-06-17T21:10:46Z","snapshot_observed_at":"2026-07-06T17:39:24.823145Z","submitted_at":"2024-03-04T18:47:08Z","title":"Taming Throughput-Latency Tradeoff in LLM Inference with Sarathi-Serve","version":3},"cited_work":{"arxiv_id":"2403.02310","doi":"10.48550/arxiv.2403.02310","metadata_source":"pith","pith_arxiv_id":"2403.02310","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gulavani, Alexey Tumanov, and Ramachandran Ramjee","venue":"cs.LG","work_id":"07bbeba4-bd20-4699-9af8-2d24435f090a","year":2024},"citing_paper":{"arxiv_id":"2606.28565","last_updated":"2026-07-02T09:24:03Z","snapshot_observed_at":"2026-08-03T10:20:30.500902Z","submitted_at":"2026-06-26T19:43:38Z","title":"KernelSight-LM: A Kernel-Level LLM Inference Simulator","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-07-03T23:09:38.092583Z"},"links":{"cited_paper":"/paper/2403.02310","citing_paper":"/paper/2606.28565"},"observation_digest":"sha256:ff90a4eab081cd43f742c95911d940adf52ab23b0aac81ad0e2d82864cc13a00","observation_id":"c25b08c1-89d8-4736-af91-0e1d640d5bc4","resolution":{"observed_at":"2026-07-03T23:19:02.267496Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.02310","last_updated":"2024-06-17T21:10:46Z","snapshot_observed_at":"2026-07-06T17:39:24.823145Z","submitted_at":"2024-03-04T18:47:08Z","title":"Taming Throughput-Latency Tradeoff in LLM Inference with Sarathi-Serve","version":3},"cited_work":{"arxiv_id":"2403.02310","doi":"10.48550/arxiv.2403.02310","metadata_source":"pith","pith_arxiv_id":"2403.02310","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gulavani, Alexey Tumanov, and Ramachandran Ramjee","venue":"cs.LG","work_id":"07bbeba4-bd20-4699-9af8-2d24435f090a","year":2024},"citing_paper":{"arxiv_id":"2606.31093","last_updated":"2026-06-30T03:29:21Z","snapshot_observed_at":"2026-08-01T17:26:20.214245Z","submitted_at":"2026-06-30T03:29:21Z","title":"Omni-Flow: A Unified Workflow Orchestration and Distributed KV Cache Sharing Framework for Multimodal Inference","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-07-01T04:13:53.722933Z"},"links":{"cited_paper":"/paper/2403.02310","citing_paper":"/paper/2606.31093"},"observation_digest":"sha256:2f06477e279ded828f7e388c54b0a60f418511696da8353c496400c8973a626e","observation_id":"d2306179-367e-4b31-a14f-2eea20d99c93","resolution":{"observed_at":"2026-07-01T11:35:43.651571Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.02310","last_updated":"2024-06-17T21:10:46Z","snapshot_observed_at":"2026-07-06T17:39:24.823145Z","submitted_at":"2024-03-04T18:47:08Z","title":"Taming Throughput-Latency Tradeoff in LLM Inference with Sarathi-Serve","version":3},"cited_work":{"arxiv_id":"2403.02310","doi":"10.48550/arxiv.2403.02310","metadata_source":"pith","pith_arxiv_id":"2403.02310","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gulavani, Alexey Tumanov, and Ramachandran Ramjee","venue":"cs.LG","work_id":"07bbeba4-bd20-4699-9af8-2d24435f090a","year":2024},"citing_paper":{"arxiv_id":"2607.01579","last_updated":"2026-07-02T01:23:02Z","snapshot_observed_at":"2026-08-02T15:54:32.616830Z","submitted_at":"2026-07-02T01:23:02Z","title":"OmniPilot: An Uncertainty-Aware LLM Inference Advisor for Heterogeneous GPU Clusters","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-07-03T06:30:30.308713Z"},"links":{"cited_paper":"/paper/2403.02310","citing_paper":"/paper/2607.01579"},"observation_digest":"sha256:1cbbade2598c52b00f409cee40ca6224e7b865998fa62c5deb505632003afe1f","observation_id":"4c37c6f3-b581-4bfd-aa59-fab3fe947a51","resolution":{"observed_at":"2026-07-03T06:37:42.164177Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.02310","last_updated":"2024-06-17T21:10:46Z","snapshot_observed_at":"2026-07-06T17:39:24.823145Z","submitted_at":"2024-03-04T18:47:08Z","title":"Taming Throughput-Latency Tradeoff in LLM Inference with Sarathi-Serve","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.02310","snapshot_observed_at":"2026-07-11T15:30:19.741437Z","title":"Taming Throughput-Latency Tradeoff in LLM Inference with Sarathi-Serve,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.04668","last_updated":"2026-07-06T04:50:32Z","snapshot_observed_at":"2026-08-02T05:19:19.975540Z","submitted_at":"2026-07-06T04:50:32Z","title":"Elastic Gang: Per-Token Membership Change for a Hard-Barriered LLM Inference Gang Co-Scheduled with OS Processes","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-07-11T15:30:19.741437Z"},"links":{"cited_paper":"/paper/2403.02310","citing_paper":"/paper/2607.04668"},"observation_digest":"sha256:b2d94b129f922b529006056d56dd680504816f63457b2603f5c8c24ba03d08cb","observation_id":"fbfecaf3-58cc-4976-a941-5373c19bff38","resolution":{"observed_at":"2026-07-11T15:30:19.741437Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.02310","last_updated":"2024-06-17T21:10:46Z","snapshot_observed_at":"2026-07-06T17:39:24.823145Z","submitted_at":"2024-03-04T18:47:08Z","title":"Taming Throughput-Latency Tradeoff in LLM Inference with Sarathi-Serve","version":3},"cited_work":{"arxiv_id":"2403.02310","doi":"10.48550/arxiv.2403.02310","metadata_source":"pith","pith_arxiv_id":"2403.02310","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gulavani, Alexey Tumanov, and Ramachandran Ramjee","venue":"cs.LG","work_id":"07bbeba4-bd20-4699-9af8-2d24435f090a","year":2024},"citing_paper":{"arxiv_id":"2607.05876","last_updated":"2026-07-08T02:47:41Z","snapshot_observed_at":"2026-08-05T07:46:13.093303Z","submitted_at":"2026-07-07T06:11:54Z","title":"Think Before You Grid-Search: Floor-First Triage for LLM Serving","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-07-08T22:38:12.637901Z"},"links":{"cited_paper":"/paper/2403.02310","citing_paper":"/paper/2607.05876"},"observation_digest":"sha256:a38e041021e74d676d5d3443ce52c961ee5fb765c1abe9797c9515b02d5096b1","observation_id":"37ddf9c1-8738-4d6a-a367-f9c2f8226ff3","resolution":{"observed_at":"2026-07-08T22:45:40.096209Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.02310","last_updated":"2024-06-17T21:10:46Z","snapshot_observed_at":"2026-07-06T17:39:24.823145Z","submitted_at":"2024-03-04T18:47:08Z","title":"Taming Throughput-Latency Tradeoff in LLM Inference with Sarathi-Serve","version":3},"cited_work":{"arxiv_id":"2403.02310","doi":"10.48550/arxiv.2403.02310","metadata_source":"pith","pith_arxiv_id":"2403.02310","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gulavani, Alexey Tumanov, and Ramachandran Ramjee","venue":"cs.LG","work_id":"07bbeba4-bd20-4699-9af8-2d24435f090a","year":2024},"citing_paper":{"arxiv_id":"2607.05876","last_updated":"2026-07-08T02:47:41Z","snapshot_observed_at":"2026-08-05T07:46:13.093303Z","submitted_at":"2026-07-07T06:11:54Z","title":"Think Before You Grid-Search: Floor-First Triage for LLM Serving","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-07-11T01:55:09.658053Z"},"links":{"cited_paper":"/paper/2403.02310","citing_paper":"/paper/2607.05876"},"observation_digest":"sha256:733ffb41083570cf62d55a6bb1a98eaadd5f02faf0ed2ee4d77e2c6bbec14c6d","observation_id":"bfd610e5-6faf-4584-9844-83ca056397f3","resolution":{"observed_at":"2026-07-11T01:57:51.380223Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.02310","last_updated":"2024-06-17T21:10:46Z","snapshot_observed_at":"2026-07-06T17:39:24.823145Z","submitted_at":"2024-03-04T18:47:08Z","title":"Taming Throughput-Latency Tradeoff in LLM Inference with Sarathi-Serve","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.02310","snapshot_observed_at":"2026-08-01T07:46:26.617631Z","title":"Taming Throughput-Latency Tradeoff in LLM Inference with Sarathi- Serve","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.21686","last_updated":"2026-07-23T14:39:46Z","snapshot_observed_at":"2026-08-03T21:25:48.155923Z","submitted_at":"2026-07-23T14:39:46Z","title":"Persistent Computational State: A Session-Centric Runtime for Generative World Models","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-01T07:46:26.617631Z"},"links":{"cited_paper":"/paper/2403.02310","citing_paper":"/paper/2607.21686"},"observation_digest":"sha256:cbb141aa3c4c829c4827cd624cf2ab2497cba7f38ab2882f5be4acbcd174c18c","observation_id":"b7863588-a8d2-43b4-8caf-9c84ee16a394","resolution":{"observed_at":"2026-08-01T07:46:26.617631Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.02310","last_updated":"2024-06-17T21:10:46Z","snapshot_observed_at":"2026-07-06T17:39:24.823145Z","submitted_at":"2024-03-04T18:47:08Z","title":"Taming Throughput-Latency Tradeoff in LLM Inference with Sarathi-Serve","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.02310","snapshot_observed_at":"2026-08-04T10:54:00.179459Z","title":"Amey Agrawal, Ashish Panwar, Jayashree Mohan, Nipun Kwatra, Bhargav S Gulavani, and Ramachandran Ramjee","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.02244","last_updated":"2026-08-03T13:55:20Z","snapshot_observed_at":"2026-08-05T13:45:48.232387Z","submitted_at":"2026-08-03T13:55:20Z","title":"Efficiency and Cost Alignment in Batched LLM Serving via Resource-Fair Scheduling","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-04T10:54:00.179459Z"},"links":{"cited_paper":"/paper/2403.02310","citing_paper":"/paper/2608.02244"},"observation_digest":"sha256:114ef1e30440d7cb014499a97f57ba1b9b9a02868159d9665e0fbe3e5dd5feb1","observation_id":"46da00a1-c551-488c-8393-79b665384216","resolution":{"observed_at":"2026-08-04T10:54:00.179459Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2403.02310/citation-record","integrity":"/paper/2403.02310/integrity","json":"/paper/2403.02310/citation-record.json","paper":"/paper/2403.02310"},"outbound":[],"paper":{"arxiv_id":"2403.02310","last_updated":"2024-06-17T21:10:46Z","latest_version":3,"primary_category":"cs.LG","snapshot_observed_at":"2026-07-06T17:39:24.823145Z","submitted_at":"2024-03-04T18:47:08Z","title":"Taming Throughput-Latency Tradeoff in LLM Inference with Sarathi-Serve"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"thesis":"As of 5 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 31 inbound Pith citation observations for arXiv:2403.02310."}