{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:JZGA25VUMUYXJ2NSWDA54F6WOR","short_pith_number":"pith:JZGA25VU","schema_version":"1.0","canonical_sha256":"4e4c0d76b4653174e9b2b0c1de17d6744b847cdfb05b6adcf9e252f4614cad65","source":{"kind":"arxiv","id":"2102.07845","version":3},"attestation_state":"computed","paper":{"title":"MARINA: Faster Non-Convex Distributed Learning with Compression","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.OC"],"primary_cat":"cs.LG","authors_text":"Eduard Gorbunov, Konstantin Burlachenko, Peter Richt\\'arik, Zhize Li","submitted_at":"2021-02-15T20:50:26Z","abstract_excerpt":"We develop and analyze MARINA: a new communication efficient method for non-convex distributed learning over heterogeneous datasets. MARINA employs a novel communication compression strategy based on the compression of gradient differences that is reminiscent of but different from the strategy employed in the DIANA method of Mishchenko et al. (2019). Unlike virtually all competing distributed first-order methods, including DIANA, ours is based on a carefully designed biased gradient estimator, which is the key to its superior theoretical and practical performance. The communication complexity "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2102.07845","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-02-15T20:50:26Z","cross_cats_sorted":["math.OC"],"title_canon_sha256":"b605f0762366f0b7fe104a6aca7ba9eef40548c8746bc501a47b3ac5cc51b869","abstract_canon_sha256":"45c949e6b1b55cf6ab27db9031070cd0acf3a6d3f5641720e576930b1a1f2699"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:46:44.088180Z","signature_b64":"gDyAyc0Crpk20KkxJQTjMrNRwbqIDB01a1yihI0VNNBC2TC7L9up/D6z7ePlCXQNPCOOQ9IQaSjlB/UeAQw1DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4e4c0d76b4653174e9b2b0c1de17d6744b847cdfb05b6adcf9e252f4614cad65","last_reissued_at":"2026-07-05T03:46:44.087742Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:46:44.087742Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MARINA: Faster Non-Convex Distributed Learning with Compression","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.OC"],"primary_cat":"cs.LG","authors_text":"Eduard Gorbunov, Konstantin Burlachenko, Peter Richt\\'arik, Zhize Li","submitted_at":"2021-02-15T20:50:26Z","abstract_excerpt":"We develop and analyze MARINA: a new communication efficient method for non-convex distributed learning over heterogeneous datasets. MARINA employs a novel communication compression strategy based on the compression of gradient differences that is reminiscent of but different from the strategy employed in the DIANA method of Mishchenko et al. (2019). Unlike virtually all competing distributed first-order methods, including DIANA, ours is based on a carefully designed biased gradient estimator, which is the key to its superior theoretical and practical performance. The communication complexity "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2102.07845","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2102.07845/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2102.07845","created_at":"2026-07-05T03:46:44.087797+00:00"},{"alias_kind":"arxiv_version","alias_value":"2102.07845v3","created_at":"2026-07-05T03:46:44.087797+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2102.07845","created_at":"2026-07-05T03:46:44.087797+00:00"},{"alias_kind":"pith_short_12","alias_value":"JZGA25VUMUYX","created_at":"2026-07-05T03:46:44.087797+00:00"},{"alias_kind":"pith_short_16","alias_value":"JZGA25VUMUYXJ2NS","created_at":"2026-07-05T03:46:44.087797+00:00"},{"alias_kind":"pith_short_8","alias_value":"JZGA25VU","created_at":"2026-07-05T03:46:44.087797+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.20866","citing_title":"LOSCAR-SGD: Local SGD with Communication-Computation Overlap and Delay-Corrected Sparse Model Averaging","ref_index":138,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18174","citing_title":"Ringmaster LMO: Asynchronous Linear Minimization Oracle Momentum Method","ref_index":136,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08871","citing_title":"Rennala MVR: Improved Time Complexity for Parallel Stochastic Optimization via Momentum-Based Variance Reduction","ref_index":134,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10689","citing_title":"Communication-Efficient Gluon in Federated Learning","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07795","citing_title":"Scalable Distributed Stochastic Optimization via Bidirectional Compression: Beyond Pessimistic Limits","ref_index":57,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JZGA25VUMUYXJ2NSWDA54F6WOR","json":"https://pith.science/pith/JZGA25VUMUYXJ2NSWDA54F6WOR.json","graph_json":"https://pith.science/api/pith-number/JZGA25VUMUYXJ2NSWDA54F6WOR/graph.json","events_json":"https://pith.science/api/pith-number/JZGA25VUMUYXJ2NSWDA54F6WOR/events.json","paper":"https://pith.science/paper/JZGA25VU"},"agent_actions":{"view_html":"https://pith.science/pith/JZGA25VUMUYXJ2NSWDA54F6WOR","download_json":"https://pith.science/pith/JZGA25VUMUYXJ2NSWDA54F6WOR.json","view_paper":"https://pith.science/paper/JZGA25VU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2102.07845&json=true","fetch_graph":"https://pith.science/api/pith-number/JZGA25VUMUYXJ2NSWDA54F6WOR/graph.json","fetch_events":"https://pith.science/api/pith-number/JZGA25VUMUYXJ2NSWDA54F6WOR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JZGA25VUMUYXJ2NSWDA54F6WOR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JZGA25VUMUYXJ2NSWDA54F6WOR/action/storage_attestation","attest_author":"https://pith.science/pith/JZGA25VUMUYXJ2NSWDA54F6WOR/action/author_attestation","sign_citation":"https://pith.science/pith/JZGA25VUMUYXJ2NSWDA54F6WOR/action/citation_signature","submit_replication":"https://pith.science/pith/JZGA25VUMUYXJ2NSWDA54F6WOR/action/replication_record"}},"created_at":"2026-07-05T03:46:44.087797+00:00","updated_at":"2026-07-05T03:46:44.087797+00:00"}