{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:R5ZVZY7RJCK5L6FWRL2BCO4ZQN","short_pith_number":"pith:R5ZVZY7R","schema_version":"1.0","canonical_sha256":"8f735ce3f14895d5f8b68af4113b998349805eb5d66c88cdb61071c2d4c15e2e","source":{"kind":"arxiv","id":"2105.07829","version":1},"attestation_state":"computed","paper":{"title":"Compressed Communication for Distributed Training: Adaptive Methods and System","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","stat.ML"],"primary_cat":"cs.DC","authors_text":"Cong Xie, Haibin Lin, Shuai Zheng, Yuchen Zhong","submitted_at":"2021-05-17T13:41:47Z","abstract_excerpt":"Communication overhead severely hinders the scalability of distributed machine learning systems. Recently, there has been a growing interest in using gradient compression to reduce the communication overhead of the distributed training. However, there is little understanding of applying gradient compression to adaptive gradient methods. Moreover, its performance benefits are often limited by the non-negligible compression overhead. In this paper, we first introduce a novel adaptive gradient method with gradient compression. We show that the proposed method has a convergence rate of $\\mathcal{O"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2105.07829","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.DC","submitted_at":"2021-05-17T13:41:47Z","cross_cats_sorted":["cs.LG","stat.ML"],"title_canon_sha256":"7bcfc430f2d2cd36876756fc47bcd9266397dfb529b56e7b51d5b782c236467f","abstract_canon_sha256":"ef98bed8deda925e7f8f3684b2a8ab8bf45a1a262fb356476c4c9a247c1da520"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:40:51.460676Z","signature_b64":"bGULJ8T86pKQAGPIzHP1jxMHQeGxDPB/9WQuJso02AHaYFvleqbyT5ajdE28H+p5kvdFI2b3xKsJfFoj2NOqBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8f735ce3f14895d5f8b68af4113b998349805eb5d66c88cdb61071c2d4c15e2e","last_reissued_at":"2026-07-05T02:40:51.460232Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:40:51.460232Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Compressed Communication for Distributed Training: Adaptive Methods and System","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","stat.ML"],"primary_cat":"cs.DC","authors_text":"Cong Xie, Haibin Lin, Shuai Zheng, Yuchen Zhong","submitted_at":"2021-05-17T13:41:47Z","abstract_excerpt":"Communication overhead severely hinders the scalability of distributed machine learning systems. Recently, there has been a growing interest in using gradient compression to reduce the communication overhead of the distributed training. However, there is little understanding of applying gradient compression to adaptive gradient methods. Moreover, its performance benefits are often limited by the non-negligible compression overhead. In this paper, we first introduce a novel adaptive gradient method with gradient compression. We show that the proposed method has a convergence rate of $\\mathcal{O"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2105.07829","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2105.07829/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2105.07829","created_at":"2026-07-05T02:40:51.460285+00:00"},{"alias_kind":"arxiv_version","alias_value":"2105.07829v1","created_at":"2026-07-05T02:40:51.460285+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2105.07829","created_at":"2026-07-05T02:40:51.460285+00:00"},{"alias_kind":"pith_short_12","alias_value":"R5ZVZY7RJCK5","created_at":"2026-07-05T02:40:51.460285+00:00"},{"alias_kind":"pith_short_16","alias_value":"R5ZVZY7RJCK5L6FW","created_at":"2026-07-05T02:40:51.460285+00:00"},{"alias_kind":"pith_short_8","alias_value":"R5ZVZY7R","created_at":"2026-07-05T02:40:51.460285+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/R5ZVZY7RJCK5L6FWRL2BCO4ZQN","json":"https://pith.science/pith/R5ZVZY7RJCK5L6FWRL2BCO4ZQN.json","graph_json":"https://pith.science/api/pith-number/R5ZVZY7RJCK5L6FWRL2BCO4ZQN/graph.json","events_json":"https://pith.science/api/pith-number/R5ZVZY7RJCK5L6FWRL2BCO4ZQN/events.json","paper":"https://pith.science/paper/R5ZVZY7R"},"agent_actions":{"view_html":"https://pith.science/pith/R5ZVZY7RJCK5L6FWRL2BCO4ZQN","download_json":"https://pith.science/pith/R5ZVZY7RJCK5L6FWRL2BCO4ZQN.json","view_paper":"https://pith.science/paper/R5ZVZY7R","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2105.07829&json=true","fetch_graph":"https://pith.science/api/pith-number/R5ZVZY7RJCK5L6FWRL2BCO4ZQN/graph.json","fetch_events":"https://pith.science/api/pith-number/R5ZVZY7RJCK5L6FWRL2BCO4ZQN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/R5ZVZY7RJCK5L6FWRL2BCO4ZQN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/R5ZVZY7RJCK5L6FWRL2BCO4ZQN/action/storage_attestation","attest_author":"https://pith.science/pith/R5ZVZY7RJCK5L6FWRL2BCO4ZQN/action/author_attestation","sign_citation":"https://pith.science/pith/R5ZVZY7RJCK5L6FWRL2BCO4ZQN/action/citation_signature","submit_replication":"https://pith.science/pith/R5ZVZY7RJCK5L6FWRL2BCO4ZQN/action/replication_record"}},"created_at":"2026-07-05T02:40:51.460285+00:00","updated_at":"2026-07-05T02:40:51.460285+00:00"}