{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:GHGZHR4SOWGX4TA3PXLCKRLDTD","short_pith_number":"pith:GHGZHR4S","schema_version":"1.0","canonical_sha256":"31cd93c792758d7e4c1b7dd625456398e6251473abcd60e225de119b96fb4e8b","source":{"kind":"arxiv","id":"2403.02310","version":3},"attestation_state":"computed","paper":{"title":"Taming Throughput-Latency Tradeoff in LLM Inference with Sarathi-Serve","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.DC"],"primary_cat":"cs.LG","authors_text":"Alexey Tumanov, Amey Agrawal, Ashish Panwar, Bhargav S. Gulavani, Jayashree Mohan, Nipun Kwatra, Nitin Kedia, Ramachandran Ramjee","submitted_at":"2024-03-04T18:47:08Z","abstract_excerpt":"Each LLM serving request goes through two phases. The first is prefill which processes the entire input prompt and produces the first output token and the second is decode which generates the rest of output tokens, one-at-a-time. Prefill iterations have high latency but saturate GPU compute due to parallel processing of the input prompt. In contrast, decode iterations have low latency but also low compute utilization because a decode iteration processes only a single token per request. This makes batching highly effective for decodes and consequently for overall throughput. However, batching m"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.02310","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-03-04T18:47:08Z","cross_cats_sorted":["cs.DC"],"title_canon_sha256":"fb6101eaa47cfb38dc7c3b00b9657dc6764af414e743aea0ff6f5b2ba154f349","abstract_canon_sha256":"61171ab1ac6d3cc06344843109ff0dde116bddafdf3beb2a4873e09ab5e48647"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:33:01.323735Z","signature_b64":"4oHdDkR4vyvEJKq3TolsEDvMMuoAkDpNokXngBuHSOHcgHKXmhyG25uV+9ujVvE0Gx4Pv2zvnxvrA9HGNVpKAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"31cd93c792758d7e4c1b7dd625456398e6251473abcd60e225de119b96fb4e8b","last_reissued_at":"2026-07-05T08:33:01.323232Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:33:01.323232Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Taming Throughput-Latency Tradeoff in LLM Inference with Sarathi-Serve","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.DC"],"primary_cat":"cs.LG","authors_text":"Alexey Tumanov, Amey Agrawal, Ashish Panwar, Bhargav S. Gulavani, Jayashree Mohan, Nipun Kwatra, Nitin Kedia, Ramachandran Ramjee","submitted_at":"2024-03-04T18:47:08Z","abstract_excerpt":"Each LLM serving request goes through two phases. The first is prefill which processes the entire input prompt and produces the first output token and the second is decode which generates the rest of output tokens, one-at-a-time. Prefill iterations have high latency but saturate GPU compute due to parallel processing of the input prompt. In contrast, decode iterations have low latency but also low compute utilization because a decode iteration processes only a single token per request. This makes batching highly effective for decodes and consequently for overall throughput. However, batching m"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.02310","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.02310/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.02310","created_at":"2026-07-05T08:33:01.323291+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.02310v3","created_at":"2026-07-05T08:33:01.323291+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.02310","created_at":"2026-07-05T08:33:01.323291+00:00"},{"alias_kind":"pith_short_12","alias_value":"GHGZHR4SOWGX","created_at":"2026-07-05T08:33:01.323291+00:00"},{"alias_kind":"pith_short_16","alias_value":"GHGZHR4SOWGX4TA3","created_at":"2026-07-05T08:33:01.323291+00:00"},{"alias_kind":"pith_short_8","alias_value":"GHGZHR4S","created_at":"2026-07-05T08:33:01.323291+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":26,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2607.05876","citing_title":"Think Before You Grid-Search: Floor-First Triage for LLM Serving","ref_index":1,"is_internal_anchor":true},{"citing_arxiv_id":"2607.05876","citing_title":"Think Before You Grid-Search: Floor-First Triage for LLM Serving","ref_index":1,"is_internal_anchor":true},{"citing_arxiv_id":"2606.25097","citing_title":"Speculative Decoding at Temperature Zero: A Scoped Safety-Invariance Screen with a 48,072-Sample Expansion","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22013","citing_title":"Load Testing for Machine Learning Model Serving Systems at Scale","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20295","citing_title":"Token-Operations-Oriented Inference Optimization Techniques for Large Models","ref_index":183,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28565","citing_title":"KernelSight-LM: A Kernel-Level LLM Inference Simulator","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18431","citing_title":"Beyond Prediction: Tail-Aware Scheduling for LLM Inference","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01579","citing_title":"OmniPilot: An Uncertainty-Aware LLM Inference Advisor for Heterogeneous GPU Clusters","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26666","citing_title":"PersistentKV: Page-Aware Decode Scheduling for Long-Context LLM Serving on Commodity GPUs","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06453","citing_title":"Vortex: Efficient and Programmable Sparse Attention Serving for AI Agents","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00735","citing_title":"ViBE: Co-Optimizing Workload Skew and Hardware Variability for MoE Serving","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31093","citing_title":"Omni-Flow: A Unified Workflow Orchestration and Distributed KV Cache Sharing Framework for Multimodal Inference","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28565","citing_title":"KernelSight-LM: A Kernel-Level LLM Inference Simulator","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25550","citing_title":"DisagFusion: Asynchronous Pipeline Parallelism and Elastic Scheduling for Disaggregated Diffusion Serving","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27763","citing_title":"A Paired Testing Protocol for Batch-Conditioned Refusal Robustness in LLM Serving","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2505.09999","citing_title":"ServeGen: Workload Characterization and Generation of Large Language Model Serving in Production","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07985","citing_title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22733","citing_title":"HarnessAPI: A Skill-First Framework for Unified Streaming APIs and MCP Tools","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17410","citing_title":"Computational Challenges in Token Economics: Bridging Economic Theory and AI System Design","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2512.09427","citing_title":"ODMA: On-Demand Memory Allocation Strategy for LLM Serving on LPDDR-Class Accelerators","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2601.14910","citing_title":"PipeWeave: Synergizing Analytical and Learning Models for Unified GPU Performance Prediction","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2404.14294","citing_title":"A Survey on Efficient Inference for Large Language Models","ref_index":283,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01214","citing_title":"Agentic AI Systems Should Be Designed as Marginal Token Allocators","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07985","citing_title":"Dooly: Configuration-Agnostic, Redundancy-Aware Profiling for LLM Inference Simulation","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17353","citing_title":"Hive: A Multi-Agent Infrastructure for Algorithm- and Task-Level Scaling","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GHGZHR4SOWGX4TA3PXLCKRLDTD","json":"https://pith.science/pith/GHGZHR4SOWGX4TA3PXLCKRLDTD.json","graph_json":"https://pith.science/api/pith-number/GHGZHR4SOWGX4TA3PXLCKRLDTD/graph.json","events_json":"https://pith.science/api/pith-number/GHGZHR4SOWGX4TA3PXLCKRLDTD/events.json","paper":"https://pith.science/paper/GHGZHR4S"},"agent_actions":{"view_html":"https://pith.science/pith/GHGZHR4SOWGX4TA3PXLCKRLDTD","download_json":"https://pith.science/pith/GHGZHR4SOWGX4TA3PXLCKRLDTD.json","view_paper":"https://pith.science/paper/GHGZHR4S","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.02310&json=true","fetch_graph":"https://pith.science/api/pith-number/GHGZHR4SOWGX4TA3PXLCKRLDTD/graph.json","fetch_events":"https://pith.science/api/pith-number/GHGZHR4SOWGX4TA3PXLCKRLDTD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GHGZHR4SOWGX4TA3PXLCKRLDTD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GHGZHR4SOWGX4TA3PXLCKRLDTD/action/storage_attestation","attest_author":"https://pith.science/pith/GHGZHR4SOWGX4TA3PXLCKRLDTD/action/author_attestation","sign_citation":"https://pith.science/pith/GHGZHR4SOWGX4TA3PXLCKRLDTD/action/citation_signature","submit_replication":"https://pith.science/pith/GHGZHR4SOWGX4TA3PXLCKRLDTD/action/replication_record"}},"created_at":"2026-07-05T08:33:01.323291+00:00","updated_at":"2026-07-05T08:33:01.323291+00:00"}