{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:FITYQVD5QZSOUHB7VEF6VNW6JU","short_pith_number":"pith:FITYQVD5","schema_version":"1.0","canonical_sha256":"2a2788547d8664ea1c3fa90beab6de4d37a02a188d9f90ea1c13014dc19e907a","source":{"kind":"arxiv","id":"2506.21901","version":1},"attestation_state":"computed","paper":{"title":"A Survey of LLM Inference Systems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.DB","authors_text":"Guoliang Li, James Pan","submitted_at":"2025-06-27T04:38:20Z","abstract_excerpt":"The past few years has witnessed specialized large language model (LLM) inference systems, such as vLLM, SGLang, Mooncake, and DeepFlow, alongside rapid LLM adoption via services like ChatGPT. Driving these system design efforts is the unique autoregressive nature of LLM request processing, motivating new techniques for achieving high performance while preserving high inference quality over high-volume and high-velocity workloads. While many of these techniques are discussed across the literature, they have not been analyzed under the framework of a complete inference system, nor have the syst"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.21901","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.DB","submitted_at":"2025-06-27T04:38:20Z","cross_cats_sorted":[],"title_canon_sha256":"4f658baa0063283ef84fec5914b5f6b7b7f576c004f8e4cea18658c96cf9c2d1","abstract_canon_sha256":"42b6088145ed58d70cc4876b3e6112706d3aedfc9afb38590c6f823f9194d7d9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:28:03.709966Z","signature_b64":"F7DpEqZNYn+pJrHwToVBg73YyTVcRytZYRice8kLbLcgEzoJNfZnHl5fdST1No5xXuYuIZEin2nWkLf3ZIyNDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2a2788547d8664ea1c3fa90beab6de4d37a02a188d9f90ea1c13014dc19e907a","last_reissued_at":"2026-07-05T11:28:03.709499Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:28:03.709499Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Survey of LLM Inference Systems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.DB","authors_text":"Guoliang Li, James Pan","submitted_at":"2025-06-27T04:38:20Z","abstract_excerpt":"The past few years has witnessed specialized large language model (LLM) inference systems, such as vLLM, SGLang, Mooncake, and DeepFlow, alongside rapid LLM adoption via services like ChatGPT. Driving these system design efforts is the unique autoregressive nature of LLM request processing, motivating new techniques for achieving high performance while preserving high inference quality over high-volume and high-velocity workloads. While many of these techniques are discussed across the literature, they have not been analyzed under the framework of a complete inference system, nor have the syst"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.21901","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.21901/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.21901","created_at":"2026-07-05T11:28:03.709563+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.21901v1","created_at":"2026-07-05T11:28:03.709563+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.21901","created_at":"2026-07-05T11:28:03.709563+00:00"},{"alias_kind":"pith_short_12","alias_value":"FITYQVD5QZSO","created_at":"2026-07-05T11:28:03.709563+00:00"},{"alias_kind":"pith_short_16","alias_value":"FITYQVD5QZSOUHB7","created_at":"2026-07-05T11:28:03.709563+00:00"},{"alias_kind":"pith_short_8","alias_value":"FITYQVD5","created_at":"2026-07-05T11:28:03.709563+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.27435","citing_title":"When NPUs Are Not Always Faster: A Stage-Level Analysis of Mobile LLM Inference","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2603.21354","citing_title":"The Workload-Router-Pool Architecture for LLM Inference Optimization: A Vision Paper from the vLLM Semantic Router Project","ref_index":107,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06548","citing_title":"Continuous Latent Diffusion Language Model","ref_index":73,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FITYQVD5QZSOUHB7VEF6VNW6JU","json":"https://pith.science/pith/FITYQVD5QZSOUHB7VEF6VNW6JU.json","graph_json":"https://pith.science/api/pith-number/FITYQVD5QZSOUHB7VEF6VNW6JU/graph.json","events_json":"https://pith.science/api/pith-number/FITYQVD5QZSOUHB7VEF6VNW6JU/events.json","paper":"https://pith.science/paper/FITYQVD5"},"agent_actions":{"view_html":"https://pith.science/pith/FITYQVD5QZSOUHB7VEF6VNW6JU","download_json":"https://pith.science/pith/FITYQVD5QZSOUHB7VEF6VNW6JU.json","view_paper":"https://pith.science/paper/FITYQVD5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.21901&json=true","fetch_graph":"https://pith.science/api/pith-number/FITYQVD5QZSOUHB7VEF6VNW6JU/graph.json","fetch_events":"https://pith.science/api/pith-number/FITYQVD5QZSOUHB7VEF6VNW6JU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FITYQVD5QZSOUHB7VEF6VNW6JU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FITYQVD5QZSOUHB7VEF6VNW6JU/action/storage_attestation","attest_author":"https://pith.science/pith/FITYQVD5QZSOUHB7VEF6VNW6JU/action/author_attestation","sign_citation":"https://pith.science/pith/FITYQVD5QZSOUHB7VEF6VNW6JU/action/citation_signature","submit_replication":"https://pith.science/pith/FITYQVD5QZSOUHB7VEF6VNW6JU/action/replication_record"}},"created_at":"2026-07-05T11:28:03.709563+00:00","updated_at":"2026-07-05T11:28:03.709563+00:00"}