{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:X6ZQYPCKQYVFKSVXLTITVOPWGY","short_pith_number":"pith:X6ZQYPCK","schema_version":"1.0","canonical_sha256":"bfb30c3c4a862a554ab75cd13ab9f636264625dfbdfc5ee46e168eebd999b907","source":{"kind":"arxiv","id":"2501.16168","version":3},"attestation_state":"computed","paper":{"title":"Ringmaster ASGD: The First Asynchronous SGD with Optimal Time Complexity","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.DC","math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Alexander Tyurin, Artavazd Maranjyan, Peter Richt\\'arik","submitted_at":"2025-01-27T16:07:26Z","abstract_excerpt":"Asynchronous Stochastic Gradient Descent (Asynchronous SGD) is a cornerstone method for parallelizing learning in distributed machine learning. However, its performance suffers under arbitrarily heterogeneous computation times across workers, leading to suboptimal time complexity and inefficiency as the number of workers scales. While several Asynchronous SGD variants have been proposed, recent findings by Tyurin & Richt\\'arik (NeurIPS 2023) reveal that none achieve optimal time complexity, leaving a significant gap in the literature. In this paper, we propose Ringmaster ASGD, a novel Asynchro"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.16168","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-01-27T16:07:26Z","cross_cats_sorted":["cs.DC","math.OC","stat.ML"],"title_canon_sha256":"38699611888e05f60f7f0afda5f28b1fcfb6118da8578127a504d8efa03a3b49","abstract_canon_sha256":"4fc4a75e8aeb0c153a5ba958f4d6fe021d94c2ff55c13c87ce5818015b54f3cb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:14:53.037489Z","signature_b64":"og+5AaQobTIUVRQSsg82w4f+MoiaT1KbrHCxZa2E2BvYF0i6SnetFSt2MS8ad1EniPZyQtND/NAJVsIcDAT7Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bfb30c3c4a862a554ab75cd13ab9f636264625dfbdfc5ee46e168eebd999b907","last_reissued_at":"2026-07-05T11:14:53.037021Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:14:53.037021Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Ringmaster ASGD: The First Asynchronous SGD with Optimal Time Complexity","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.DC","math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Alexander Tyurin, Artavazd Maranjyan, Peter Richt\\'arik","submitted_at":"2025-01-27T16:07:26Z","abstract_excerpt":"Asynchronous Stochastic Gradient Descent (Asynchronous SGD) is a cornerstone method for parallelizing learning in distributed machine learning. However, its performance suffers under arbitrarily heterogeneous computation times across workers, leading to suboptimal time complexity and inefficiency as the number of workers scales. While several Asynchronous SGD variants have been proposed, recent findings by Tyurin & Richt\\'arik (NeurIPS 2023) reveal that none achieve optimal time complexity, leaving a significant gap in the literature. In this paper, we propose Ringmaster ASGD, a novel Asynchro"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.16168","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.16168/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.16168","created_at":"2026-07-05T11:14:53.037079+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.16168v3","created_at":"2026-07-05T11:14:53.037079+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.16168","created_at":"2026-07-05T11:14:53.037079+00:00"},{"alias_kind":"pith_short_12","alias_value":"X6ZQYPCKQYVF","created_at":"2026-07-05T11:14:53.037079+00:00"},{"alias_kind":"pith_short_16","alias_value":"X6ZQYPCKQYVFKSVX","created_at":"2026-07-05T11:14:53.037079+00:00"},{"alias_kind":"pith_short_8","alias_value":"X6ZQYPCK","created_at":"2026-07-05T11:14:53.037079+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.30634","citing_title":"One-Step Gradient Delay is Not a Barrier for Large-Scale Asynchronous Pipeline Parallel LLM Pretraining","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02043","citing_title":"Bringing Order to Asynchronous SGD: Towards Optimality under Data-Dependent Delays with Momentum","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02043","citing_title":"Bringing Order to Asynchronous SGD: Towards Optimality under Data-Dependent Delays with Momentum","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/X6ZQYPCKQYVFKSVXLTITVOPWGY","json":"https://pith.science/pith/X6ZQYPCKQYVFKSVXLTITVOPWGY.json","graph_json":"https://pith.science/api/pith-number/X6ZQYPCKQYVFKSVXLTITVOPWGY/graph.json","events_json":"https://pith.science/api/pith-number/X6ZQYPCKQYVFKSVXLTITVOPWGY/events.json","paper":"https://pith.science/paper/X6ZQYPCK"},"agent_actions":{"view_html":"https://pith.science/pith/X6ZQYPCKQYVFKSVXLTITVOPWGY","download_json":"https://pith.science/pith/X6ZQYPCKQYVFKSVXLTITVOPWGY.json","view_paper":"https://pith.science/paper/X6ZQYPCK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.16168&json=true","fetch_graph":"https://pith.science/api/pith-number/X6ZQYPCKQYVFKSVXLTITVOPWGY/graph.json","fetch_events":"https://pith.science/api/pith-number/X6ZQYPCKQYVFKSVXLTITVOPWGY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/X6ZQYPCKQYVFKSVXLTITVOPWGY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/X6ZQYPCKQYVFKSVXLTITVOPWGY/action/storage_attestation","attest_author":"https://pith.science/pith/X6ZQYPCKQYVFKSVXLTITVOPWGY/action/author_attestation","sign_citation":"https://pith.science/pith/X6ZQYPCKQYVFKSVXLTITVOPWGY/action/citation_signature","submit_replication":"https://pith.science/pith/X6ZQYPCKQYVFKSVXLTITVOPWGY/action/replication_record"}},"created_at":"2026-07-05T11:14:53.037079+00:00","updated_at":"2026-07-05T11:14:53.037079+00:00"}