{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:FI5VC2F3VFEJWRMUSZY2QP4I7R","short_pith_number":"pith:FI5VC2F3","schema_version":"1.0","canonical_sha256":"2a3b5168bba9489b45949671a83f88fc47a9fcad485b0d12bd09cab7edb27ce6","source":{"kind":"arxiv","id":"2504.20828","version":2},"attestation_state":"computed","paper":{"title":"Ascendra: Dynamic Request Prioritization for Efficient LLM Serving","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Azam Ikram, Sameh Elnikety, Saurabh Bagchi, Xiang Li","submitted_at":"2025-04-29T14:51:26Z","abstract_excerpt":"The rapid advancement of Large Language Models (LLMs) has driven the need for more efficient serving strategies. In this context, efficiency refers to the proportion of requests that meet their Service Level Objectives (SLOs), particularly for Time To First Token (TTFT) and Time Between Tokens (TBT). However, existing systems often prioritize one metric at the cost of the other. We present Ascendra, an LLM serving system designed to meet both TTFT and TBT SLOs simultaneously. The core insight behind Ascendra is that a request's urgency evolves as it approaches its deadline. To leverage this, A"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.20828","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-04-29T14:51:26Z","cross_cats_sorted":[],"title_canon_sha256":"46965b8b5ee79aebedbc58e2bb4fcecb22f0eec54193cd78c1024006428d804a","abstract_canon_sha256":"cf55a9c6d6705325db7a5c606d472e1fc58517fd8b6349afcca0ae444e56199e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:56:19.111234Z","signature_b64":"8D62SgzhpSoh6xsCGp7GiNEBjNvZ7uy3j780HyPn+JguwKeTtEQETTffRX6SnQvUySyoKaUdyviws3oARklPCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2a3b5168bba9489b45949671a83f88fc47a9fcad485b0d12bd09cab7edb27ce6","last_reissued_at":"2026-07-05T10:56:19.110746Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:56:19.110746Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Ascendra: Dynamic Request Prioritization for Efficient LLM Serving","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Azam Ikram, Sameh Elnikety, Saurabh Bagchi, Xiang Li","submitted_at":"2025-04-29T14:51:26Z","abstract_excerpt":"The rapid advancement of Large Language Models (LLMs) has driven the need for more efficient serving strategies. In this context, efficiency refers to the proportion of requests that meet their Service Level Objectives (SLOs), particularly for Time To First Token (TTFT) and Time Between Tokens (TBT). However, existing systems often prioritize one metric at the cost of the other. We present Ascendra, an LLM serving system designed to meet both TTFT and TBT SLOs simultaneously. The core insight behind Ascendra is that a request's urgency evolves as it approaches its deadline. To leverage this, A"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.20828","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.20828/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.20828","created_at":"2026-07-05T10:56:19.110805+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.20828v2","created_at":"2026-07-05T10:56:19.110805+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.20828","created_at":"2026-07-05T10:56:19.110805+00:00"},{"alias_kind":"pith_short_12","alias_value":"FI5VC2F3VFEJ","created_at":"2026-07-05T10:56:19.110805+00:00"},{"alias_kind":"pith_short_16","alias_value":"FI5VC2F3VFEJWRMU","created_at":"2026-07-05T10:56:19.110805+00:00"},{"alias_kind":"pith_short_8","alias_value":"FI5VC2F3","created_at":"2026-07-05T10:56:19.110805+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.02329","citing_title":"Taming Request Imbalance: SLO-Aware Scheduling for Disaggregated LLM Inference","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2601.21841","citing_title":"Embodied Task Planning via Graph-Informed Action Generation with Large Language Models","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02329","citing_title":"Taming Request Imbalance: SLO-Aware Scheduling for Disaggregated LLM Inference","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FI5VC2F3VFEJWRMUSZY2QP4I7R","json":"https://pith.science/pith/FI5VC2F3VFEJWRMUSZY2QP4I7R.json","graph_json":"https://pith.science/api/pith-number/FI5VC2F3VFEJWRMUSZY2QP4I7R/graph.json","events_json":"https://pith.science/api/pith-number/FI5VC2F3VFEJWRMUSZY2QP4I7R/events.json","paper":"https://pith.science/paper/FI5VC2F3"},"agent_actions":{"view_html":"https://pith.science/pith/FI5VC2F3VFEJWRMUSZY2QP4I7R","download_json":"https://pith.science/pith/FI5VC2F3VFEJWRMUSZY2QP4I7R.json","view_paper":"https://pith.science/paper/FI5VC2F3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.20828&json=true","fetch_graph":"https://pith.science/api/pith-number/FI5VC2F3VFEJWRMUSZY2QP4I7R/graph.json","fetch_events":"https://pith.science/api/pith-number/FI5VC2F3VFEJWRMUSZY2QP4I7R/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FI5VC2F3VFEJWRMUSZY2QP4I7R/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FI5VC2F3VFEJWRMUSZY2QP4I7R/action/storage_attestation","attest_author":"https://pith.science/pith/FI5VC2F3VFEJWRMUSZY2QP4I7R/action/author_attestation","sign_citation":"https://pith.science/pith/FI5VC2F3VFEJWRMUSZY2QP4I7R/action/citation_signature","submit_replication":"https://pith.science/pith/FI5VC2F3VFEJWRMUSZY2QP4I7R/action/replication_record"}},"created_at":"2026-07-05T10:56:19.110805+00:00","updated_at":"2026-07-05T10:56:19.110805+00:00"}