{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:K3LOUZZRII7HUPXJ3SOMCBZGJJ","short_pith_number":"pith:K3LOUZZR","schema_version":"1.0","canonical_sha256":"56d6ea6731423e7a3ee9dc9cc107264a6c0306d434e5e4eb64aca843300f97ef","source":{"kind":"arxiv","id":"2504.14775","version":2},"attestation_state":"computed","paper":{"title":"gLLM: Global Balanced Pipeline Parallelism System for Distributed LLM Serving with Token Throttling","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.DC","authors_text":"Jiangsu Du, Nong Xiao, Tianyu Guo, Xianwei Zhang, Yutong Lu, Zhiguang Chen","submitted_at":"2025-04-21T00:07:49Z","abstract_excerpt":"Pipeline parallelism has emerged as a predominant approach for deploying large language models (LLMs) across distributed nodes, owing to its lower communication overhead compared to tensor parallelism. While demonstrating high throughput in request serving, pipeline parallelism often suffers from performance limitations caused by pipeline bubbles, which are primarily resulted from imbalanced computation delays across batches. Existing methods like Sarathi-Serve attempt to address this through hybrid scheduling of chunked prefill and decode tokens using a fixed token budget. However, such metho"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.14775","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.DC","submitted_at":"2025-04-21T00:07:49Z","cross_cats_sorted":[],"title_canon_sha256":"0792f048f6e5ae14d38eb870433a0c121daca10b9cf9f9f934eb9a709ce21732","abstract_canon_sha256":"fa8ae8c4b28e061c59b9f7ae6010ec1daed86f9b64a3c1649401cc6e39a63677"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:10:56.024206Z","signature_b64":"4ReM6AyQGHuyzAUNhpqm+cW8hQVuMl5LRPPFiHTw5SXw+DBn1diwbUYTfEf0l298NYVSMJUwyR5Xfb0IFgBVDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"56d6ea6731423e7a3ee9dc9cc107264a6c0306d434e5e4eb64aca843300f97ef","last_reissued_at":"2026-07-05T11:10:56.023562Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:10:56.023562Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"gLLM: Global Balanced Pipeline Parallelism System for Distributed LLM Serving with Token Throttling","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.DC","authors_text":"Jiangsu Du, Nong Xiao, Tianyu Guo, Xianwei Zhang, Yutong Lu, Zhiguang Chen","submitted_at":"2025-04-21T00:07:49Z","abstract_excerpt":"Pipeline parallelism has emerged as a predominant approach for deploying large language models (LLMs) across distributed nodes, owing to its lower communication overhead compared to tensor parallelism. While demonstrating high throughput in request serving, pipeline parallelism often suffers from performance limitations caused by pipeline bubbles, which are primarily resulted from imbalanced computation delays across batches. Existing methods like Sarathi-Serve attempt to address this through hybrid scheduling of chunked prefill and decode tokens using a fixed token budget. However, such metho"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.14775","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.14775/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.14775","created_at":"2026-07-05T11:10:56.023680+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.14775v2","created_at":"2026-07-05T11:10:56.023680+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.14775","created_at":"2026-07-05T11:10:56.023680+00:00"},{"alias_kind":"pith_short_12","alias_value":"K3LOUZZRII7H","created_at":"2026-07-05T11:10:56.023680+00:00"},{"alias_kind":"pith_short_16","alias_value":"K3LOUZZRII7HUPXJ","created_at":"2026-07-05T11:10:56.023680+00:00"},{"alias_kind":"pith_short_8","alias_value":"K3LOUZZR","created_at":"2026-07-05T11:10:56.023680+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.22033","citing_title":"SiPipe: Bridging the CPU-GPU Utilization Gap for Efficient Pipeline-Parallel LLM Inference","ref_index":20,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/K3LOUZZRII7HUPXJ3SOMCBZGJJ","json":"https://pith.science/pith/K3LOUZZRII7HUPXJ3SOMCBZGJJ.json","graph_json":"https://pith.science/api/pith-number/K3LOUZZRII7HUPXJ3SOMCBZGJJ/graph.json","events_json":"https://pith.science/api/pith-number/K3LOUZZRII7HUPXJ3SOMCBZGJJ/events.json","paper":"https://pith.science/paper/K3LOUZZR"},"agent_actions":{"view_html":"https://pith.science/pith/K3LOUZZRII7HUPXJ3SOMCBZGJJ","download_json":"https://pith.science/pith/K3LOUZZRII7HUPXJ3SOMCBZGJJ.json","view_paper":"https://pith.science/paper/K3LOUZZR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.14775&json=true","fetch_graph":"https://pith.science/api/pith-number/K3LOUZZRII7HUPXJ3SOMCBZGJJ/graph.json","fetch_events":"https://pith.science/api/pith-number/K3LOUZZRII7HUPXJ3SOMCBZGJJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/K3LOUZZRII7HUPXJ3SOMCBZGJJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/K3LOUZZRII7HUPXJ3SOMCBZGJJ/action/storage_attestation","attest_author":"https://pith.science/pith/K3LOUZZRII7HUPXJ3SOMCBZGJJ/action/author_attestation","sign_citation":"https://pith.science/pith/K3LOUZZRII7HUPXJ3SOMCBZGJJ/action/citation_signature","submit_replication":"https://pith.science/pith/K3LOUZZRII7HUPXJ3SOMCBZGJJ/action/replication_record"}},"created_at":"2026-07-05T11:10:56.023680+00:00","updated_at":"2026-07-05T11:10:56.023680+00:00"}