{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:N2KQMU2DH52A5KNZALRIBDJYEL","short_pith_number":"pith:N2KQMU2D","schema_version":"1.0","canonical_sha256":"6e950653433f740ea9b902e2808d3822ead30d952917372c8a486754137897cf","source":{"kind":"arxiv","id":"2607.23250","version":1},"attestation_state":"computed","paper":{"title":"Libra: Taming Attention Workload Skew in Long-Context LLM Training with Bounded Sequence Pool","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.DC","authors_text":"Chang Si, Guangming Tan, Hongqing Chen, Jiaxuan Peng, Jingren Zhou, Kaiming Yang, Langshi Chen, Linlang Jiang, Man Yuan, Mingzhen Li, Pengju Lu, Rui Men, Siyu Wang, Weile Jia, Xiulong Yuan, Yan Wang, Yong Li, Zhipeng Zhang, Zhixiang Ruan","submitted_at":"2026-07-25T15:21:25Z","abstract_excerpt":"Long-context LLM training suffers from a load-balancing problem that sequence packing does not solve. Packing samples into fixed-token sequences balances memory and linear-cost operators, but the dominant attention cost scales with the sum of squared sequence lengths. Thus, equally sized packed sequences drawn from a long-tailed corpus can carry substantially different attention workloads, creating data-parallel stragglers and pipeline bubbles. Existing approaches either balance at the granularity of sequences or microbatches, where an outlier can dominate an assignment, or disaggregate attent"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.23250","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.DC","submitted_at":"2026-07-25T15:21:25Z","cross_cats_sorted":[],"title_canon_sha256":"cee256284de67edc67d968c8c4bb5d83855fdcab2ad5d6600d338591dc0ccc7b","abstract_canon_sha256":"01c93b93a06b2d944a35285a6e323982f653ecb7059e55634705bf92e3b7f270"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-28T01:22:44.017308Z","signature_b64":"iWh8DaTTal1Q/p3WGEGByFSrOZSDX/RqTYj0YrAyq5+ldAG/l4KX93uqNrWEMDVO9ZP+X1i3tjBP3B1aDJByCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6e950653433f740ea9b902e2808d3822ead30d952917372c8a486754137897cf","last_reissued_at":"2026-07-28T01:22:44.016452Z","signature_status":"signed_v1","first_computed_at":"2026-07-28T01:22:44.016452Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Libra: Taming Attention Workload Skew in Long-Context LLM Training with Bounded Sequence Pool","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.DC","authors_text":"Chang Si, Guangming Tan, Hongqing Chen, Jiaxuan Peng, Jingren Zhou, Kaiming Yang, Langshi Chen, Linlang Jiang, Man Yuan, Mingzhen Li, Pengju Lu, Rui Men, Siyu Wang, Weile Jia, Xiulong Yuan, Yan Wang, Yong Li, Zhipeng Zhang, Zhixiang Ruan","submitted_at":"2026-07-25T15:21:25Z","abstract_excerpt":"Long-context LLM training suffers from a load-balancing problem that sequence packing does not solve. Packing samples into fixed-token sequences balances memory and linear-cost operators, but the dominant attention cost scales with the sum of squared sequence lengths. Thus, equally sized packed sequences drawn from a long-tailed corpus can carry substantially different attention workloads, creating data-parallel stragglers and pipeline bubbles. Existing approaches either balance at the granularity of sequences or microbatches, where an outlier can dominate an assignment, or disaggregate attent"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.23250","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.23250/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.23250","created_at":"2026-07-28T01:22:44.016887+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.23250v1","created_at":"2026-07-28T01:22:44.016887+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.23250","created_at":"2026-07-28T01:22:44.016887+00:00"},{"alias_kind":"pith_short_12","alias_value":"N2KQMU2DH52A","created_at":"2026-07-28T01:22:44.016887+00:00"},{"alias_kind":"pith_short_16","alias_value":"N2KQMU2DH52A5KNZ","created_at":"2026-07-28T01:22:44.016887+00:00"},{"alias_kind":"pith_short_8","alias_value":"N2KQMU2D","created_at":"2026-07-28T01:22:44.016887+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/N2KQMU2DH52A5KNZALRIBDJYEL","json":"https://pith.science/pith/N2KQMU2DH52A5KNZALRIBDJYEL.json","graph_json":"https://pith.science/api/pith-number/N2KQMU2DH52A5KNZALRIBDJYEL/graph.json","events_json":"https://pith.science/api/pith-number/N2KQMU2DH52A5KNZALRIBDJYEL/events.json","paper":"https://pith.science/paper/N2KQMU2D"},"agent_actions":{"view_html":"https://pith.science/pith/N2KQMU2DH52A5KNZALRIBDJYEL","download_json":"https://pith.science/pith/N2KQMU2DH52A5KNZALRIBDJYEL.json","view_paper":"https://pith.science/paper/N2KQMU2D","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.23250&json=true","fetch_graph":"https://pith.science/api/pith-number/N2KQMU2DH52A5KNZALRIBDJYEL/graph.json","fetch_events":"https://pith.science/api/pith-number/N2KQMU2DH52A5KNZALRIBDJYEL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/N2KQMU2DH52A5KNZALRIBDJYEL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/N2KQMU2DH52A5KNZALRIBDJYEL/action/storage_attestation","attest_author":"https://pith.science/pith/N2KQMU2DH52A5KNZALRIBDJYEL/action/author_attestation","sign_citation":"https://pith.science/pith/N2KQMU2DH52A5KNZALRIBDJYEL/action/citation_signature","submit_replication":"https://pith.science/pith/N2KQMU2DH52A5KNZALRIBDJYEL/action/replication_record"}},"created_at":"2026-07-28T01:22:44.016887+00:00","updated_at":"2026-07-28T01:22:44.016887+00:00"}