{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:HXZVNGVNANN5K7Y7ES2MGXQKAY","short_pith_number":"pith:HXZVNGVN","schema_version":"1.0","canonical_sha256":"3df3569aad035bd57f1f24b4c35e0a06248583f7e6ebd49991ae6ee9bbe1d27c","source":{"kind":"arxiv","id":"2503.05248","version":1},"attestation_state":"computed","paper":{"title":"Optimizing LLM Inference Throughput via Memory-aware and SLA-constrained Dynamic Batching","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.DC","authors_text":"Bowen Pang, Feifan Wang, Kai Li","submitted_at":"2025-03-07T09:02:21Z","abstract_excerpt":"The increasing adoption of large language models (LLMs) necessitates inference serving systems that can deliver both high throughput and low latency. Deploying LLMs with hundreds of billions of parameters on memory-constrained GPUs exposes significant limitations in static batching methods. Current inference serving systems often treat batch sizes as fixed hyper-parameters, hindering real-time adaptation to varying system conditions. In this paper, we propose a dynamic batching method that continuously monitors memory utilization and adheres to service-level agreements (SLAs) to enable real-ti"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.05248","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.DC","submitted_at":"2025-03-07T09:02:21Z","cross_cats_sorted":[],"title_canon_sha256":"d08230d8a316973fa6d9ba5e4a7e79460b43358ae6c1855efacc2f210d17395c","abstract_canon_sha256":"1540861508c9f3dbce0ee6957d88d0a3b9039791f11720a1ade5e2fb697a8044"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:26:24.892381Z","signature_b64":"hd5MpMvy8ERRAjoPyfgP4k7Q3JkPOY0IXYGW1XbKNdC3vlO3Qq2aL/BDuA7LVg6k00ONPc35vNQyoQZZ4dkJAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3df3569aad035bd57f1f24b4c35e0a06248583f7e6ebd49991ae6ee9bbe1d27c","last_reissued_at":"2026-07-05T10:26:24.891883Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:26:24.891883Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Optimizing LLM Inference Throughput via Memory-aware and SLA-constrained Dynamic Batching","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.DC","authors_text":"Bowen Pang, Feifan Wang, Kai Li","submitted_at":"2025-03-07T09:02:21Z","abstract_excerpt":"The increasing adoption of large language models (LLMs) necessitates inference serving systems that can deliver both high throughput and low latency. Deploying LLMs with hundreds of billions of parameters on memory-constrained GPUs exposes significant limitations in static batching methods. Current inference serving systems often treat batch sizes as fixed hyper-parameters, hindering real-time adaptation to varying system conditions. In this paper, we propose a dynamic batching method that continuously monitors memory utilization and adheres to service-level agreements (SLAs) to enable real-ti"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.05248","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.05248/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.05248","created_at":"2026-07-05T10:26:24.891944+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.05248v1","created_at":"2026-07-05T10:26:24.891944+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.05248","created_at":"2026-07-05T10:26:24.891944+00:00"},{"alias_kind":"pith_short_12","alias_value":"HXZVNGVNANN5","created_at":"2026-07-05T10:26:24.891944+00:00"},{"alias_kind":"pith_short_16","alias_value":"HXZVNGVNANN5K7Y7","created_at":"2026-07-05T10:26:24.891944+00:00"},{"alias_kind":"pith_short_8","alias_value":"HXZVNGVN","created_at":"2026-07-05T10:26:24.891944+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22327","citing_title":"Geometry-Aware Online Scheduling for LLM Serving: From Theoretical Bound to System Practice","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06046","citing_title":"Requests of a Feather Must Flock Together: Batch Size vs. Prefix Homogeneity in LLM Inference","ref_index":30,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HXZVNGVNANN5K7Y7ES2MGXQKAY","json":"https://pith.science/pith/HXZVNGVNANN5K7Y7ES2MGXQKAY.json","graph_json":"https://pith.science/api/pith-number/HXZVNGVNANN5K7Y7ES2MGXQKAY/graph.json","events_json":"https://pith.science/api/pith-number/HXZVNGVNANN5K7Y7ES2MGXQKAY/events.json","paper":"https://pith.science/paper/HXZVNGVN"},"agent_actions":{"view_html":"https://pith.science/pith/HXZVNGVNANN5K7Y7ES2MGXQKAY","download_json":"https://pith.science/pith/HXZVNGVNANN5K7Y7ES2MGXQKAY.json","view_paper":"https://pith.science/paper/HXZVNGVN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.05248&json=true","fetch_graph":"https://pith.science/api/pith-number/HXZVNGVNANN5K7Y7ES2MGXQKAY/graph.json","fetch_events":"https://pith.science/api/pith-number/HXZVNGVNANN5K7Y7ES2MGXQKAY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HXZVNGVNANN5K7Y7ES2MGXQKAY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HXZVNGVNANN5K7Y7ES2MGXQKAY/action/storage_attestation","attest_author":"https://pith.science/pith/HXZVNGVNANN5K7Y7ES2MGXQKAY/action/author_attestation","sign_citation":"https://pith.science/pith/HXZVNGVNANN5K7Y7ES2MGXQKAY/action/citation_signature","submit_replication":"https://pith.science/pith/HXZVNGVNANN5K7Y7ES2MGXQKAY/action/replication_record"}},"created_at":"2026-07-05T10:26:24.891944+00:00","updated_at":"2026-07-05T10:26:24.891944+00:00"}