{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:RFBSY5ZVH4UOEZOX7QXL7PD6M5","short_pith_number":"pith:RFBSY5ZV","schema_version":"1.0","canonical_sha256":"89432c77353f28e265d7fc2ebfbc7e675849b9691b6dbb86a3899e9eb57ab4df","source":{"kind":"arxiv","id":"2110.08133","version":1},"attestation_state":"computed","paper":{"title":"Trade-offs of Local SGD at Scale: An Empirical Study","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.LG","authors_text":"Ari Morcos, Jonathan Frankle, Jose Javier Gonzalez Ortiz, Mike Rabbat, Nicolas Ballas","submitted_at":"2021-10-15T15:00:42Z","abstract_excerpt":"As datasets and models become increasingly large, distributed training has become a necessary component to allow deep neural networks to train in reasonable amounts of time. However, distributed training can have substantial communication overhead that hinders its scalability. One strategy for reducing this overhead is to perform multiple unsynchronized SGD steps independently on each worker between synchronization steps, a technique known as local SGD. We conduct a comprehensive empirical study of local SGD and related methods on a large-scale image classification task. We find that performin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2110.08133","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2021-10-15T15:00:42Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"24bad43d3c195d8d3e40b02c2f6125157c4e5006a3579de02f5ad85d0a7b3fb0","abstract_canon_sha256":"d2fce05a1d25dd5709c85e997f2b2b84f48ec14d2c7622b0b04b20ee8ecaec06"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:23:00.816982Z","signature_b64":"4wAmzn4Ewiup4aSpiZyZSsr/HqBOSNp2AMlZZT0VTT4fbt1RCQa4XHsGRmeZfcHXe3ZZIEg2RS4jR8wBsjCIDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"89432c77353f28e265d7fc2ebfbc7e675849b9691b6dbb86a3899e9eb57ab4df","last_reissued_at":"2026-07-05T03:23:00.816529Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:23:00.816529Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Trade-offs of Local SGD at Scale: An Empirical Study","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.LG","authors_text":"Ari Morcos, Jonathan Frankle, Jose Javier Gonzalez Ortiz, Mike Rabbat, Nicolas Ballas","submitted_at":"2021-10-15T15:00:42Z","abstract_excerpt":"As datasets and models become increasingly large, distributed training has become a necessary component to allow deep neural networks to train in reasonable amounts of time. However, distributed training can have substantial communication overhead that hinders its scalability. One strategy for reducing this overhead is to perform multiple unsynchronized SGD steps independently on each worker between synchronization steps, a technique known as local SGD. We conduct a comprehensive empirical study of local SGD and related methods on a large-scale image classification task. We find that performin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2110.08133","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2110.08133/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2110.08133","created_at":"2026-07-05T03:23:00.816590+00:00"},{"alias_kind":"arxiv_version","alias_value":"2110.08133v1","created_at":"2026-07-05T03:23:00.816590+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2110.08133","created_at":"2026-07-05T03:23:00.816590+00:00"},{"alias_kind":"pith_short_12","alias_value":"RFBSY5ZVH4UO","created_at":"2026-07-05T03:23:00.816590+00:00"},{"alias_kind":"pith_short_16","alias_value":"RFBSY5ZVH4UOEZOX","created_at":"2026-07-05T03:23:00.816590+00:00"},{"alias_kind":"pith_short_8","alias_value":"RFBSY5ZV","created_at":"2026-07-05T03:23:00.816590+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.15706","citing_title":"Overcoming the Communication-Performance Tradeoff in LLM Pretraining","ref_index":18,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RFBSY5ZVH4UOEZOX7QXL7PD6M5","json":"https://pith.science/pith/RFBSY5ZVH4UOEZOX7QXL7PD6M5.json","graph_json":"https://pith.science/api/pith-number/RFBSY5ZVH4UOEZOX7QXL7PD6M5/graph.json","events_json":"https://pith.science/api/pith-number/RFBSY5ZVH4UOEZOX7QXL7PD6M5/events.json","paper":"https://pith.science/paper/RFBSY5ZV"},"agent_actions":{"view_html":"https://pith.science/pith/RFBSY5ZVH4UOEZOX7QXL7PD6M5","download_json":"https://pith.science/pith/RFBSY5ZVH4UOEZOX7QXL7PD6M5.json","view_paper":"https://pith.science/paper/RFBSY5ZV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2110.08133&json=true","fetch_graph":"https://pith.science/api/pith-number/RFBSY5ZVH4UOEZOX7QXL7PD6M5/graph.json","fetch_events":"https://pith.science/api/pith-number/RFBSY5ZVH4UOEZOX7QXL7PD6M5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RFBSY5ZVH4UOEZOX7QXL7PD6M5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RFBSY5ZVH4UOEZOX7QXL7PD6M5/action/storage_attestation","attest_author":"https://pith.science/pith/RFBSY5ZVH4UOEZOX7QXL7PD6M5/action/author_attestation","sign_citation":"https://pith.science/pith/RFBSY5ZVH4UOEZOX7QXL7PD6M5/action/citation_signature","submit_replication":"https://pith.science/pith/RFBSY5ZVH4UOEZOX7QXL7PD6M5/action/replication_record"}},"created_at":"2026-07-05T03:23:00.816590+00:00","updated_at":"2026-07-05T03:23:00.816590+00:00"}