{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:CMK6WGMKRN2ADSWNSMV7BWACHM","short_pith_number":"pith:CMK6WGMK","schema_version":"1.0","canonical_sha256":"1315eb198a8b7401cacd932bf0d8023b26561f6f243f9b5144d519cfa98351b2","source":{"kind":"arxiv","id":"2405.06856","version":1},"attestation_state":"computed","paper":{"title":"Aladdin: Joint Placement and Scaling for SLO-Aware LLM Serving","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.DC","authors_text":"Chengyi Nie, Rodrigo Fonseca, Zhenhua Liu","submitted_at":"2024-05-11T00:00:23Z","abstract_excerpt":"The demand for large language model (LLM) inference is gradually dominating the artificial intelligence workloads. Therefore, there is an urgent need for cost-efficient inference serving. Existing work focuses on single-worker optimization and lacks consideration of cluster-level management for both inference queries and computing resources. However, placing requests and managing resources without considering the query features easily causes SLO violations or resource underutilization. Providers are forced to allocate extra computing resources to guarantee user experience, leading to additiona"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.06856","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.DC","submitted_at":"2024-05-11T00:00:23Z","cross_cats_sorted":[],"title_canon_sha256":"4398872340efc61c8ebafc3e809fc5a9ebe641e54d038fd6c97c8a514bd57d45","abstract_canon_sha256":"100b70f94fa9722546f367dbe919b525a45931120b780ab111ba4b1638c27482"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:18:12.785471Z","signature_b64":"tdI5JbfZcYzrEWq6hI8MHb44U4YsRQ+xEaTu37Jc0yZNmv78ODkWyvzFJX1WN1TEAUa9yGcSmhpLLAIOoNRlDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1315eb198a8b7401cacd932bf0d8023b26561f6f243f9b5144d519cfa98351b2","last_reissued_at":"2026-07-05T08:18:12.784998Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:18:12.784998Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Aladdin: Joint Placement and Scaling for SLO-Aware LLM Serving","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.DC","authors_text":"Chengyi Nie, Rodrigo Fonseca, Zhenhua Liu","submitted_at":"2024-05-11T00:00:23Z","abstract_excerpt":"The demand for large language model (LLM) inference is gradually dominating the artificial intelligence workloads. Therefore, there is an urgent need for cost-efficient inference serving. Existing work focuses on single-worker optimization and lacks consideration of cluster-level management for both inference queries and computing resources. However, placing requests and managing resources without considering the query features easily causes SLO violations or resource underutilization. Providers are forced to allocate extra computing resources to guarantee user experience, leading to additiona"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.06856","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.06856/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.06856","created_at":"2026-07-05T08:18:12.785056+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.06856v1","created_at":"2026-07-05T08:18:12.785056+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.06856","created_at":"2026-07-05T08:18:12.785056+00:00"},{"alias_kind":"pith_short_12","alias_value":"CMK6WGMKRN2A","created_at":"2026-07-05T08:18:12.785056+00:00"},{"alias_kind":"pith_short_16","alias_value":"CMK6WGMKRN2ADSWN","created_at":"2026-07-05T08:18:12.785056+00:00"},{"alias_kind":"pith_short_8","alias_value":"CMK6WGMK","created_at":"2026-07-05T08:18:12.785056+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2508.15919","citing_title":"HFX: Joint Design of Algorithms and Systems for Multi-SLO Serving and Fast Scaling","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2602.18755","citing_title":"DualScale: Energy-Efficient Disaggregated LLM Serving via Phase-Aware Placement and DVFS","ref_index":31,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CMK6WGMKRN2ADSWNSMV7BWACHM","json":"https://pith.science/pith/CMK6WGMKRN2ADSWNSMV7BWACHM.json","graph_json":"https://pith.science/api/pith-number/CMK6WGMKRN2ADSWNSMV7BWACHM/graph.json","events_json":"https://pith.science/api/pith-number/CMK6WGMKRN2ADSWNSMV7BWACHM/events.json","paper":"https://pith.science/paper/CMK6WGMK"},"agent_actions":{"view_html":"https://pith.science/pith/CMK6WGMKRN2ADSWNSMV7BWACHM","download_json":"https://pith.science/pith/CMK6WGMKRN2ADSWNSMV7BWACHM.json","view_paper":"https://pith.science/paper/CMK6WGMK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.06856&json=true","fetch_graph":"https://pith.science/api/pith-number/CMK6WGMKRN2ADSWNSMV7BWACHM/graph.json","fetch_events":"https://pith.science/api/pith-number/CMK6WGMKRN2ADSWNSMV7BWACHM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CMK6WGMKRN2ADSWNSMV7BWACHM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CMK6WGMKRN2ADSWNSMV7BWACHM/action/storage_attestation","attest_author":"https://pith.science/pith/CMK6WGMKRN2ADSWNSMV7BWACHM/action/author_attestation","sign_citation":"https://pith.science/pith/CMK6WGMKRN2ADSWNSMV7BWACHM/action/citation_signature","submit_replication":"https://pith.science/pith/CMK6WGMKRN2ADSWNSMV7BWACHM/action/replication_record"}},"created_at":"2026-07-05T08:18:12.785056+00:00","updated_at":"2026-07-05T08:18:12.785056+00:00"}