{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:SIBYRZFKLPENNGKPKWKJZE2R5E","short_pith_number":"pith:SIBYRZFK","schema_version":"1.0","canonical_sha256":"920388e4aa5bc8d6994f55949c9351e90875a2e4ba483c4927fce6e222cd7a21","source":{"kind":"arxiv","id":"2501.08090","version":1},"attestation_state":"computed","paper":{"title":"Hierarchical Autoscaling for Large Language Model Serving with Chiron","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.DC","authors_text":"Archit Patke, Chandra Narayanaswami, Dhemath Reddy, Ravishankar Iyer, Saurabh Jha, Zbigniew Kalbarczyk","submitted_at":"2025-01-14T12:57:40Z","abstract_excerpt":"Large language model (LLM) serving is becoming an increasingly important workload for cloud providers. Based on performance SLO requirements, LLM inference requests can be divided into (a) interactive requests that have tight SLOs in the order of seconds, and (b) batch requests that have relaxed SLO in the order of minutes to hours. These SLOs can degrade based on the arrival rates, multiplexing, and configuration parameters, thus necessitating the use of resource autoscaling on serving instances and their batch sizes. However, previous autoscalers for LLM serving do not consider request SLOs "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.08090","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.DC","submitted_at":"2025-01-14T12:57:40Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"df786d3336904d7c22741baeee8f57aa884c3ca4bd6acccc6092a7bcad97a291","abstract_canon_sha256":"361dd5b1ee353ea0f41682f3e14242a41c70e7640767b8d776bfa91be69cbd70"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:00:55.579562Z","signature_b64":"9L2TMTltupWTh0n4S4Jk3Y7PjjC/mLQ7rd5ncfFyRtZp49ADHNN/O83Ao/gW2oIJdnEhNtzcQ8CMoFjXwDEhDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"920388e4aa5bc8d6994f55949c9351e90875a2e4ba483c4927fce6e222cd7a21","last_reissued_at":"2026-07-05T10:00:55.578907Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:00:55.578907Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Hierarchical Autoscaling for Large Language Model Serving with Chiron","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.DC","authors_text":"Archit Patke, Chandra Narayanaswami, Dhemath Reddy, Ravishankar Iyer, Saurabh Jha, Zbigniew Kalbarczyk","submitted_at":"2025-01-14T12:57:40Z","abstract_excerpt":"Large language model (LLM) serving is becoming an increasingly important workload for cloud providers. Based on performance SLO requirements, LLM inference requests can be divided into (a) interactive requests that have tight SLOs in the order of seconds, and (b) batch requests that have relaxed SLO in the order of minutes to hours. These SLOs can degrade based on the arrival rates, multiplexing, and configuration parameters, thus necessitating the use of resource autoscaling on serving instances and their batch sizes. However, previous autoscalers for LLM serving do not consider request SLOs "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.08090","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.08090/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.08090","created_at":"2026-07-05T10:00:55.578968+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.08090v1","created_at":"2026-07-05T10:00:55.578968+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.08090","created_at":"2026-07-05T10:00:55.578968+00:00"},{"alias_kind":"pith_short_12","alias_value":"SIBYRZFKLPEN","created_at":"2026-07-05T10:00:55.578968+00:00"},{"alias_kind":"pith_short_16","alias_value":"SIBYRZFKLPENNGKP","created_at":"2026-07-05T10:00:55.578968+00:00"},{"alias_kind":"pith_short_8","alias_value":"SIBYRZFK","created_at":"2026-07-05T10:00:55.578968+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.05467","citing_title":"Nitsum: Serving Tiered LLM Requests with Adaptive Tensor Parallelism","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16682","citing_title":"KAIROS: Stateful, Context-Aware Power-Efficient Agentic Inference Serving","ref_index":49,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SIBYRZFKLPENNGKPKWKJZE2R5E","json":"https://pith.science/pith/SIBYRZFKLPENNGKPKWKJZE2R5E.json","graph_json":"https://pith.science/api/pith-number/SIBYRZFKLPENNGKPKWKJZE2R5E/graph.json","events_json":"https://pith.science/api/pith-number/SIBYRZFKLPENNGKPKWKJZE2R5E/events.json","paper":"https://pith.science/paper/SIBYRZFK"},"agent_actions":{"view_html":"https://pith.science/pith/SIBYRZFKLPENNGKPKWKJZE2R5E","download_json":"https://pith.science/pith/SIBYRZFKLPENNGKPKWKJZE2R5E.json","view_paper":"https://pith.science/paper/SIBYRZFK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.08090&json=true","fetch_graph":"https://pith.science/api/pith-number/SIBYRZFKLPENNGKPKWKJZE2R5E/graph.json","fetch_events":"https://pith.science/api/pith-number/SIBYRZFKLPENNGKPKWKJZE2R5E/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SIBYRZFKLPENNGKPKWKJZE2R5E/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SIBYRZFKLPENNGKPKWKJZE2R5E/action/storage_attestation","attest_author":"https://pith.science/pith/SIBYRZFKLPENNGKPKWKJZE2R5E/action/author_attestation","sign_citation":"https://pith.science/pith/SIBYRZFKLPENNGKPKWKJZE2R5E/action/citation_signature","submit_replication":"https://pith.science/pith/SIBYRZFKLPENNGKPKWKJZE2R5E/action/replication_record"}},"created_at":"2026-07-05T10:00:55.578968+00:00","updated_at":"2026-07-05T10:00:55.578968+00:00"}