{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:KRYYZY36TDH6OX7GXWDX5B33TP","short_pith_number":"pith:KRYYZY36","schema_version":"1.0","canonical_sha256":"54718ce37e98cfe75fe6bd877e877b9bc1e9885be14adc8d19a5bfa4bbe67bcf","source":{"kind":"arxiv","id":"2401.00588","version":2},"attestation_state":"computed","paper":{"title":"Fairness in Serving Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.PF"],"primary_cat":"cs.AI","authors_text":"Banghua Zhu, Dacheng Li, Danyang Zhuo, Ion Stoica, Joseph E. Gonzalez, Shiyi Cao, Ying Sheng, Zhuohan Li","submitted_at":"2023-12-31T21:15:54Z","abstract_excerpt":"High-demand LLM inference services (e.g., ChatGPT and BARD) support a wide range of requests from short chat conversations to long document reading. To ensure that all client requests are processed fairly, most major LLM inference services have request rate limits, to ensure that no client can dominate the request queue. However, this rudimentary notion of fairness also results in under-utilization of the resources and poor client experience when there is spare capacity. While there is a rich literature on fair scheduling, serving LLMs presents new challenges due to their unpredictable request"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.00588","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2023-12-31T21:15:54Z","cross_cats_sorted":["cs.LG","cs.PF"],"title_canon_sha256":"eaa02262dc2e392cdb7e360fd236ef831467fe75001d8a9c44d0325448eeb2b7","abstract_canon_sha256":"64593f3be966ed1fc8215d5407b048e15beb7c729f5726f74366dd4ef33a7c2e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:27:24.275124Z","signature_b64":"Lce3YqHePsSd94UKR30aiaPYpQW9qIZDF0BlHUfcrc7TJTLVGQk8KvQTtPrWer49JcOT5hdcp/j8YMHMrjWsDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"54718ce37e98cfe75fe6bd877e877b9bc1e9885be14adc8d19a5bfa4bbe67bcf","last_reissued_at":"2026-07-05T08:27:24.274675Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:27:24.274675Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Fairness in Serving Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.PF"],"primary_cat":"cs.AI","authors_text":"Banghua Zhu, Dacheng Li, Danyang Zhuo, Ion Stoica, Joseph E. Gonzalez, Shiyi Cao, Ying Sheng, Zhuohan Li","submitted_at":"2023-12-31T21:15:54Z","abstract_excerpt":"High-demand LLM inference services (e.g., ChatGPT and BARD) support a wide range of requests from short chat conversations to long document reading. To ensure that all client requests are processed fairly, most major LLM inference services have request rate limits, to ensure that no client can dominate the request queue. However, this rudimentary notion of fairness also results in under-utilization of the resources and poor client experience when there is spare capacity. While there is a rich literature on fair scheduling, serving LLMs presents new challenges due to their unpredictable request"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.00588","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.00588/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.00588","created_at":"2026-07-05T08:27:24.274729+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.00588v2","created_at":"2026-07-05T08:27:24.274729+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.00588","created_at":"2026-07-05T08:27:24.274729+00:00"},{"alias_kind":"pith_short_12","alias_value":"KRYYZY36TDH6","created_at":"2026-07-05T08:27:24.274729+00:00"},{"alias_kind":"pith_short_16","alias_value":"KRYYZY36TDH6OX7G","created_at":"2026-07-05T08:27:24.274729+00:00"},{"alias_kind":"pith_short_8","alias_value":"KRYYZY36","created_at":"2026-07-05T08:27:24.274729+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2404.14294","citing_title":"A Survey on Efficient Inference for Large Language Models","ref_index":281,"is_internal_anchor":false},{"citing_arxiv_id":"2312.07104","citing_title":"SGLang: Efficient Execution of Structured Language Model Programs","ref_index":42,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KRYYZY36TDH6OX7GXWDX5B33TP","json":"https://pith.science/pith/KRYYZY36TDH6OX7GXWDX5B33TP.json","graph_json":"https://pith.science/api/pith-number/KRYYZY36TDH6OX7GXWDX5B33TP/graph.json","events_json":"https://pith.science/api/pith-number/KRYYZY36TDH6OX7GXWDX5B33TP/events.json","paper":"https://pith.science/paper/KRYYZY36"},"agent_actions":{"view_html":"https://pith.science/pith/KRYYZY36TDH6OX7GXWDX5B33TP","download_json":"https://pith.science/pith/KRYYZY36TDH6OX7GXWDX5B33TP.json","view_paper":"https://pith.science/paper/KRYYZY36","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.00588&json=true","fetch_graph":"https://pith.science/api/pith-number/KRYYZY36TDH6OX7GXWDX5B33TP/graph.json","fetch_events":"https://pith.science/api/pith-number/KRYYZY36TDH6OX7GXWDX5B33TP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KRYYZY36TDH6OX7GXWDX5B33TP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KRYYZY36TDH6OX7GXWDX5B33TP/action/storage_attestation","attest_author":"https://pith.science/pith/KRYYZY36TDH6OX7GXWDX5B33TP/action/author_attestation","sign_citation":"https://pith.science/pith/KRYYZY36TDH6OX7GXWDX5B33TP/action/citation_signature","submit_replication":"https://pith.science/pith/KRYYZY36TDH6OX7GXWDX5B33TP/action/replication_record"}},"created_at":"2026-07-05T08:27:24.274729+00:00","updated_at":"2026-07-05T08:27:24.274729+00:00"}