{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:D3JCBPZ5A5I3ULPLAZEYHXAOOS","short_pith_number":"pith:D3JCBPZ5","schema_version":"1.0","canonical_sha256":"1ed220bf3d0751ba2deb064983dc0e748a89efd77a73eaac4a8942eaafe8add7","source":{"kind":"arxiv","id":"2412.04504","version":1},"attestation_state":"computed","paper":{"title":"Multi-Bin Batching for Increasing LLM Inference Throughput","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.DC","cs.LG","cs.SY","eess.SY"],"primary_cat":"cs.CL","authors_text":"Jackson Kunde, Kangwook Lee, Ozgur Guldogan, Ramtin Pedarsani","submitted_at":"2024-12-03T03:16:12Z","abstract_excerpt":"As large language models (LLMs) grow in popularity for their diverse capabilities, improving the efficiency of their inference systems has become increasingly critical. Batching LLM requests is a critical step in scheduling the inference jobs on servers (e.g. GPUs), enabling the system to maximize throughput by allowing multiple requests to be processed in parallel. However, requests often have varying generation lengths, causing resource underutilization, as hardware must wait for the longest-running request in the batch to complete before moving to the next batch. We formalize this problem f"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.04504","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-12-03T03:16:12Z","cross_cats_sorted":["cs.DC","cs.LG","cs.SY","eess.SY"],"title_canon_sha256":"fdd390199113ff7bff1b1bc557026397e048800e75bd1a4093358aeceae00867","abstract_canon_sha256":"ba4d6118e74e300acbd794fcdb42dba309967ab7fa435f69996d17649bb38dd7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:45:18.706608Z","signature_b64":"dqHtZ7E7NXMxoDTAxy8jn5l2AZ0HASKdVf/tvu5Owz0ZJCQDtYiU3/7s4MsJZcVnVaVNAIxQfJjNR9dK7m3TBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1ed220bf3d0751ba2deb064983dc0e748a89efd77a73eaac4a8942eaafe8add7","last_reissued_at":"2026-07-05T09:45:18.705975Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:45:18.705975Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Multi-Bin Batching for Increasing LLM Inference Throughput","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.DC","cs.LG","cs.SY","eess.SY"],"primary_cat":"cs.CL","authors_text":"Jackson Kunde, Kangwook Lee, Ozgur Guldogan, Ramtin Pedarsani","submitted_at":"2024-12-03T03:16:12Z","abstract_excerpt":"As large language models (LLMs) grow in popularity for their diverse capabilities, improving the efficiency of their inference systems has become increasingly critical. Batching LLM requests is a critical step in scheduling the inference jobs on servers (e.g. GPUs), enabling the system to maximize throughput by allowing multiple requests to be processed in parallel. However, requests often have varying generation lengths, causing resource underutilization, as hardware must wait for the longest-running request in the batch to complete before moving to the next batch. We formalize this problem f"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.04504","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.04504/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.04504","created_at":"2026-07-05T09:45:18.706052+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.04504v1","created_at":"2026-07-05T09:45:18.706052+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.04504","created_at":"2026-07-05T09:45:18.706052+00:00"},{"alias_kind":"pith_short_12","alias_value":"D3JCBPZ5A5I3","created_at":"2026-07-05T09:45:18.706052+00:00"},{"alias_kind":"pith_short_16","alias_value":"D3JCBPZ5A5I3ULPL","created_at":"2026-07-05T09:45:18.706052+00:00"},{"alias_kind":"pith_short_8","alias_value":"D3JCBPZ5","created_at":"2026-07-05T09:45:18.706052+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.04595","citing_title":"A Queueing-Theoretic Framework for Stability Analysis of LLM Inference with KV Cache Memory Constraints","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/D3JCBPZ5A5I3ULPLAZEYHXAOOS","json":"https://pith.science/pith/D3JCBPZ5A5I3ULPLAZEYHXAOOS.json","graph_json":"https://pith.science/api/pith-number/D3JCBPZ5A5I3ULPLAZEYHXAOOS/graph.json","events_json":"https://pith.science/api/pith-number/D3JCBPZ5A5I3ULPLAZEYHXAOOS/events.json","paper":"https://pith.science/paper/D3JCBPZ5"},"agent_actions":{"view_html":"https://pith.science/pith/D3JCBPZ5A5I3ULPLAZEYHXAOOS","download_json":"https://pith.science/pith/D3JCBPZ5A5I3ULPLAZEYHXAOOS.json","view_paper":"https://pith.science/paper/D3JCBPZ5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.04504&json=true","fetch_graph":"https://pith.science/api/pith-number/D3JCBPZ5A5I3ULPLAZEYHXAOOS/graph.json","fetch_events":"https://pith.science/api/pith-number/D3JCBPZ5A5I3ULPLAZEYHXAOOS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/D3JCBPZ5A5I3ULPLAZEYHXAOOS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/D3JCBPZ5A5I3ULPLAZEYHXAOOS/action/storage_attestation","attest_author":"https://pith.science/pith/D3JCBPZ5A5I3ULPLAZEYHXAOOS/action/author_attestation","sign_citation":"https://pith.science/pith/D3JCBPZ5A5I3ULPLAZEYHXAOOS/action/citation_signature","submit_replication":"https://pith.science/pith/D3JCBPZ5A5I3ULPLAZEYHXAOOS/action/replication_record"}},"created_at":"2026-07-05T09:45:18.706052+00:00","updated_at":"2026-07-05T09:45:18.706052+00:00"}