{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ZAQL5NQKMO3PX7JNFSEQKCBQTH","short_pith_number":"pith:ZAQL5NQK","schema_version":"1.0","canonical_sha256":"c820beb60a63b6fbfd2d2c8905083099e23c0169690e94d032bbc00732f19f19","source":{"kind":"arxiv","id":"2411.15997","version":1},"attestation_state":"computed","paper":{"title":"Ensuring Fair LLM Serving Amid Diverse Applications","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.DC","cs.MA"],"primary_cat":"cs.LG","authors_text":"Ali R. Butt, Anjaly Parayil, Ankur Mallick, Anoop Kulkarni, Chetan Bansal, Haiying Shen, Kunal Jain, Pankhuri Choudhary, Redwan Ibne Seraj Khan, Ren\\`ee St. Amant, Rujia Wang, Saravan Rajmohan, Steve Kofsky, Victor R\\\"uhle, Yue Cheng","submitted_at":"2024-11-24T22:35:44Z","abstract_excerpt":"In a multi-tenant large language model (LLM) serving platform hosting diverse applications, some users may submit an excessive number of requests, causing the service to become unavailable to other users and creating unfairness. Existing fairness approaches do not account for variations in token lengths across applications and multiple LLM calls, making them unsuitable for such platforms. To address the fairness challenge, this paper analyzes millions of requests from thousands of users on MS CoPilot, a real-world multi-tenant LLM platform hosted by Microsoft. Our analysis confirms the inadequ"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.15997","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-11-24T22:35:44Z","cross_cats_sorted":["cs.AI","cs.DC","cs.MA"],"title_canon_sha256":"98c9c283c29cf0d52c1f276cc562384eccfa644fa8ca1166a1e9a28def36631b","abstract_canon_sha256":"93601230cf0ef15eba5a4673f43795dcb14ea88f6421c6d2b8961a208dbdb585"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:40:02.181184Z","signature_b64":"gyLd8xd87LwWz9/O1AQFDCOIERwcbapLO3I8kbvtEDIOJtrjqLyoYTv3dOUUsupea7LP7Gv/fPhbqpu3Ybu3CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c820beb60a63b6fbfd2d2c8905083099e23c0169690e94d032bbc00732f19f19","last_reissued_at":"2026-07-05T09:40:02.180719Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:40:02.180719Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Ensuring Fair LLM Serving Amid Diverse Applications","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.DC","cs.MA"],"primary_cat":"cs.LG","authors_text":"Ali R. Butt, Anjaly Parayil, Ankur Mallick, Anoop Kulkarni, Chetan Bansal, Haiying Shen, Kunal Jain, Pankhuri Choudhary, Redwan Ibne Seraj Khan, Ren\\`ee St. Amant, Rujia Wang, Saravan Rajmohan, Steve Kofsky, Victor R\\\"uhle, Yue Cheng","submitted_at":"2024-11-24T22:35:44Z","abstract_excerpt":"In a multi-tenant large language model (LLM) serving platform hosting diverse applications, some users may submit an excessive number of requests, causing the service to become unavailable to other users and creating unfairness. Existing fairness approaches do not account for variations in token lengths across applications and multiple LLM calls, making them unsuitable for such platforms. To address the fairness challenge, this paper analyzes millions of requests from thousands of users on MS CoPilot, a real-world multi-tenant LLM platform hosted by Microsoft. Our analysis confirms the inadequ"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.15997","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.15997/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.15997","created_at":"2026-07-05T09:40:02.180777+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.15997v1","created_at":"2026-07-05T09:40:02.180777+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.15997","created_at":"2026-07-05T09:40:02.180777+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZAQL5NQKMO3P","created_at":"2026-07-05T09:40:02.180777+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZAQL5NQKMO3PX7JN","created_at":"2026-07-05T09:40:02.180777+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZAQL5NQK","created_at":"2026-07-05T09:40:02.180777+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.17227","citing_title":"Cloud-native and Distributed Systems for Efficient and Scalable Large Language Models -- A Research Agenda","ref_index":150,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZAQL5NQKMO3PX7JNFSEQKCBQTH","json":"https://pith.science/pith/ZAQL5NQKMO3PX7JNFSEQKCBQTH.json","graph_json":"https://pith.science/api/pith-number/ZAQL5NQKMO3PX7JNFSEQKCBQTH/graph.json","events_json":"https://pith.science/api/pith-number/ZAQL5NQKMO3PX7JNFSEQKCBQTH/events.json","paper":"https://pith.science/paper/ZAQL5NQK"},"agent_actions":{"view_html":"https://pith.science/pith/ZAQL5NQKMO3PX7JNFSEQKCBQTH","download_json":"https://pith.science/pith/ZAQL5NQKMO3PX7JNFSEQKCBQTH.json","view_paper":"https://pith.science/paper/ZAQL5NQK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.15997&json=true","fetch_graph":"https://pith.science/api/pith-number/ZAQL5NQKMO3PX7JNFSEQKCBQTH/graph.json","fetch_events":"https://pith.science/api/pith-number/ZAQL5NQKMO3PX7JNFSEQKCBQTH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZAQL5NQKMO3PX7JNFSEQKCBQTH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZAQL5NQKMO3PX7JNFSEQKCBQTH/action/storage_attestation","attest_author":"https://pith.science/pith/ZAQL5NQKMO3PX7JNFSEQKCBQTH/action/author_attestation","sign_citation":"https://pith.science/pith/ZAQL5NQKMO3PX7JNFSEQKCBQTH/action/citation_signature","submit_replication":"https://pith.science/pith/ZAQL5NQKMO3PX7JNFSEQKCBQTH/action/replication_record"}},"created_at":"2026-07-05T09:40:02.180777+00:00","updated_at":"2026-07-05T09:40:02.180777+00:00"}