{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:EVDXDRGX2MG7ZV7GDYADD54Y6C","short_pith_number":"pith:EVDXDRGX","schema_version":"1.0","canonical_sha256":"254771c4d7d30dfcd7e61e0031f798f08f42f6da791a677f2c96826b7bd54449","source":{"kind":"arxiv","id":"2503.09799","version":1},"attestation_state":"computed","paper":{"title":"Communication-Efficient Language Model Training Scales Reliably and Robustly: Scaling Laws for DiLoCo","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.DC"],"primary_cat":"cs.LG","authors_text":"Arthur Douillard, Arthur Szlam, Gabriel Teston, Keith Rush, Lucio Dery, Nova Fallen, Zachary Charles, Zachary Garrett","submitted_at":"2025-03-12T20:04:38Z","abstract_excerpt":"As we scale to more massive machine learning models, the frequent synchronization demands inherent in data-parallel approaches create significant slowdowns, posing a critical challenge to further scaling. Recent work develops an approach (DiLoCo) that relaxes synchronization demands without compromising model quality. However, these works do not carefully analyze how DiLoCo's behavior changes with model size. In this work, we study the scaling law behavior of DiLoCo when training LLMs under a fixed compute budget. We focus on how algorithmic factors, including number of model replicas, hyperpa"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.09799","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-03-12T20:04:38Z","cross_cats_sorted":["cs.CL","cs.DC"],"title_canon_sha256":"ae3d4af5f0d0a479e820d2076c378e91e0246a6bc83e2fcc16def0d4567e33a3","abstract_canon_sha256":"b3830c4a7668feaa45c2d88519c8fec4321cda1a1adf5243eb9dab7681308be6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:30:08.666563Z","signature_b64":"LFe7c9RF0UlBDd0aTU2LQvpzjhTisS0vGR5ThGv7//rcxgfL1nVtjsb04GD6BQhQRHMhW0iKWpTU9PzHhIFjBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"254771c4d7d30dfcd7e61e0031f798f08f42f6da791a677f2c96826b7bd54449","last_reissued_at":"2026-07-05T10:30:08.666010Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:30:08.666010Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Communication-Efficient Language Model Training Scales Reliably and Robustly: Scaling Laws for DiLoCo","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.DC"],"primary_cat":"cs.LG","authors_text":"Arthur Douillard, Arthur Szlam, Gabriel Teston, Keith Rush, Lucio Dery, Nova Fallen, Zachary Charles, Zachary Garrett","submitted_at":"2025-03-12T20:04:38Z","abstract_excerpt":"As we scale to more massive machine learning models, the frequent synchronization demands inherent in data-parallel approaches create significant slowdowns, posing a critical challenge to further scaling. Recent work develops an approach (DiLoCo) that relaxes synchronization demands without compromising model quality. However, these works do not carefully analyze how DiLoCo's behavior changes with model size. In this work, we study the scaling law behavior of DiLoCo when training LLMs under a fixed compute budget. We focus on how algorithmic factors, including number of model replicas, hyperpa"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.09799","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.09799/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.09799","created_at":"2026-07-05T10:30:08.666087+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.09799v1","created_at":"2026-07-05T10:30:08.666087+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.09799","created_at":"2026-07-05T10:30:08.666087+00:00"},{"alias_kind":"pith_short_12","alias_value":"EVDXDRGX2MG7","created_at":"2026-07-05T10:30:08.666087+00:00"},{"alias_kind":"pith_short_16","alias_value":"EVDXDRGX2MG7ZV7G","created_at":"2026-07-05T10:30:08.666087+00:00"},{"alias_kind":"pith_short_8","alias_value":"EVDXDRGX","created_at":"2026-07-05T10:30:08.666087+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.02958","citing_title":"Echelon: Auditable Aggregate-Only Language-Model Adaptation Across Privacy Boundaries","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29359","citing_title":"Does Distributed Training Undermine Compute Governance?","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28585","citing_title":"Outer-Momentum Restarting in High-Dimensional Two-Phase Optimization","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EVDXDRGX2MG7ZV7GDYADD54Y6C","json":"https://pith.science/pith/EVDXDRGX2MG7ZV7GDYADD54Y6C.json","graph_json":"https://pith.science/api/pith-number/EVDXDRGX2MG7ZV7GDYADD54Y6C/graph.json","events_json":"https://pith.science/api/pith-number/EVDXDRGX2MG7ZV7GDYADD54Y6C/events.json","paper":"https://pith.science/paper/EVDXDRGX"},"agent_actions":{"view_html":"https://pith.science/pith/EVDXDRGX2MG7ZV7GDYADD54Y6C","download_json":"https://pith.science/pith/EVDXDRGX2MG7ZV7GDYADD54Y6C.json","view_paper":"https://pith.science/paper/EVDXDRGX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.09799&json=true","fetch_graph":"https://pith.science/api/pith-number/EVDXDRGX2MG7ZV7GDYADD54Y6C/graph.json","fetch_events":"https://pith.science/api/pith-number/EVDXDRGX2MG7ZV7GDYADD54Y6C/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EVDXDRGX2MG7ZV7GDYADD54Y6C/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EVDXDRGX2MG7ZV7GDYADD54Y6C/action/storage_attestation","attest_author":"https://pith.science/pith/EVDXDRGX2MG7ZV7GDYADD54Y6C/action/author_attestation","sign_citation":"https://pith.science/pith/EVDXDRGX2MG7ZV7GDYADD54Y6C/action/citation_signature","submit_replication":"https://pith.science/pith/EVDXDRGX2MG7ZV7GDYADD54Y6C/action/replication_record"}},"created_at":"2026-07-05T10:30:08.666087+00:00","updated_at":"2026-07-05T10:30:08.666087+00:00"}