{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:AO6X2YC4ZRATM34XH425RR3IU2","short_pith_number":"pith:AO6X2YC4","schema_version":"1.0","canonical_sha256":"03bd7d605ccc41366f973f35d8c768a68e21c646ab6febb901b6d45422bcca80","source":{"kind":"arxiv","id":"2404.14527","version":4},"attestation_state":"computed","paper":{"title":"M\\'elange: Cost Efficient Large Language Model Serving by Exploiting GPU Heterogeneity","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.DC","authors_text":"Alvin Cheung, Doyoung Kim, Ion Stoica, Jiaxiang Yu, Tyler Griggs, Wei-Lin Chiang, Xiaoxuan Liu","submitted_at":"2024-04-22T18:56:18Z","abstract_excerpt":"Large language models (LLMs) are increasingly integrated into many online services, yet they remain cost-prohibitive to deploy due to the requirement of expensive GPU instances. Prior work has addressed the high cost of LLM serving by improving the inference engine, but less attention has been given to selecting the most cost-efficient GPU type(s) for a specific LLM service. There is a large and growing landscape of GPU types and, within these options, higher cost does not always lead to increased performance. Instead, through a comprehensive investigation, we find that three key LLM service c"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.14527","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.DC","submitted_at":"2024-04-22T18:56:18Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"ab36325166569bef72eab1cb336afea03b0a40df302d0ed314a243d6c86ff60e","abstract_canon_sha256":"ac92c1dc844b6d537646902c4f3e067516d3ae0b9a1ad457df2394b927449620"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:46:54.677590Z","signature_b64":"7wJLBdPlFsQppiiDZhI5C2q3aXCfOcEz2XoZWuWb199zF2CdQp1h9vmKbioYVN8WH35w0o/XsnjaExE6YxAnCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"03bd7d605ccc41366f973f35d8c768a68e21c646ab6febb901b6d45422bcca80","last_reissued_at":"2026-07-05T08:46:54.676994Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:46:54.676994Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"M\\'elange: Cost Efficient Large Language Model Serving by Exploiting GPU Heterogeneity","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.DC","authors_text":"Alvin Cheung, Doyoung Kim, Ion Stoica, Jiaxiang Yu, Tyler Griggs, Wei-Lin Chiang, Xiaoxuan Liu","submitted_at":"2024-04-22T18:56:18Z","abstract_excerpt":"Large language models (LLMs) are increasingly integrated into many online services, yet they remain cost-prohibitive to deploy due to the requirement of expensive GPU instances. Prior work has addressed the high cost of LLM serving by improving the inference engine, but less attention has been given to selecting the most cost-efficient GPU type(s) for a specific LLM service. There is a large and growing landscape of GPU types and, within these options, higher cost does not always lead to increased performance. Instead, through a comprehensive investigation, we find that three key LLM service c"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.14527","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.14527/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.14527","created_at":"2026-07-05T08:46:54.677056+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.14527v4","created_at":"2026-07-05T08:46:54.677056+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.14527","created_at":"2026-07-05T08:46:54.677056+00:00"},{"alias_kind":"pith_short_12","alias_value":"AO6X2YC4ZRAT","created_at":"2026-07-05T08:46:54.677056+00:00"},{"alias_kind":"pith_short_16","alias_value":"AO6X2YC4ZRATM34X","created_at":"2026-07-05T08:46:54.677056+00:00"},{"alias_kind":"pith_short_8","alias_value":"AO6X2YC4","created_at":"2026-07-05T08:46:54.677056+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11690","citing_title":"Beyond Per-Token Pricing: A Concurrency-Aware Methodology for LLM Infrastructure Cost Estimation","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01839","citing_title":"Observation, Not Prediction: Conversation-Level Disaggregated Scheduling for Agentic Serving","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00866","citing_title":"Idleness is Relative: Exploiting Tool-Call Idle Windows for Offloading in Agentic Systems with MORI","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20629","citing_title":"Specialize Roles, Mix Deployments: Pushing the Cost-Accuracy Frontier of LLM Agent Teams","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19481","citing_title":"C2CServe: Leveraging NVLink-C2C for Elastic Serverless LLM Serving on MIG","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16867","citing_title":"GoodServe: Towards High-Goodput Serving of Agentic LLM Inferences over Heterogeneous Resources","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2602.09725","citing_title":"Efficient Remote KV Cache Reuse with GPU-native Video Codec","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2603.21354","citing_title":"The Workload-Router-Pool Architecture for LLM Inference Optimization: A Vision Paper from the vLLM Semantic Router Project","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04357","citing_title":"Coral: Cost-Efficient Multi-LLM Serving over Heterogeneous Cloud GPUs","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12171","citing_title":"PipeLive: Efficient Live In-place Pipeline Parallelism Reconfiguration for Dynamic LLM Serving","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AO6X2YC4ZRATM34XH425RR3IU2","json":"https://pith.science/pith/AO6X2YC4ZRATM34XH425RR3IU2.json","graph_json":"https://pith.science/api/pith-number/AO6X2YC4ZRATM34XH425RR3IU2/graph.json","events_json":"https://pith.science/api/pith-number/AO6X2YC4ZRATM34XH425RR3IU2/events.json","paper":"https://pith.science/paper/AO6X2YC4"},"agent_actions":{"view_html":"https://pith.science/pith/AO6X2YC4ZRATM34XH425RR3IU2","download_json":"https://pith.science/pith/AO6X2YC4ZRATM34XH425RR3IU2.json","view_paper":"https://pith.science/paper/AO6X2YC4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.14527&json=true","fetch_graph":"https://pith.science/api/pith-number/AO6X2YC4ZRATM34XH425RR3IU2/graph.json","fetch_events":"https://pith.science/api/pith-number/AO6X2YC4ZRATM34XH425RR3IU2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AO6X2YC4ZRATM34XH425RR3IU2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AO6X2YC4ZRATM34XH425RR3IU2/action/storage_attestation","attest_author":"https://pith.science/pith/AO6X2YC4ZRATM34XH425RR3IU2/action/author_attestation","sign_citation":"https://pith.science/pith/AO6X2YC4ZRATM34XH425RR3IU2/action/citation_signature","submit_replication":"https://pith.science/pith/AO6X2YC4ZRATM34XH425RR3IU2/action/replication_record"}},"created_at":"2026-07-05T08:46:54.677056+00:00","updated_at":"2026-07-05T08:46:54.677056+00:00"}