{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:5TYV2MO3O2GRJVMWVDAE4BBAWI","short_pith_number":"pith:5TYV2MO3","schema_version":"1.0","canonical_sha256":"ecf15d31db768d14d596a8c04e0420b23a082db5af841404c55586f748820ed0","source":{"kind":"arxiv","id":"2002.12410","version":4},"attestation_state":"computed","paper":{"title":"On Biased Compression for Distributed Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.DC","math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Aleksandr Beznosikov, Mher Safaryan, Peter Richt\\'arik, Samuel Horv\\'ath","submitted_at":"2020-02-27T19:52:24Z","abstract_excerpt":"In the last few years, various communication compression techniques have emerged as an indispensable tool helping to alleviate the communication bottleneck in distributed learning. However, despite the fact biased compressors often show superior performance in practice when compared to the much more studied and understood unbiased compressors, very little is known about them. In this work we study three classes of biased compression operators, two of which are new, and their performance when applied to (stochastic) gradient descent and distributed (stochastic) gradient descent. We show for the"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2002.12410","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-02-27T19:52:24Z","cross_cats_sorted":["cs.DC","math.OC","stat.ML"],"title_canon_sha256":"9234037e2d3a2859a158e53046c39e0eba79cad44c13d5f28dc31c010f985ffb","abstract_canon_sha256":"8214466387d6c6fe33085c3b7df44694853bbbf883078e8ed21af22a13709f95"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:33:05.554600Z","signature_b64":"pK5ja3puhQhxSDf+YIEGqLsOAR0WESdnqEW16uuqjCkXQEQbzwAKXqI9eCKk3rNYVn4uEcpKdUUvj+AP/MVGBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ecf15d31db768d14d596a8c04e0420b23a082db5af841404c55586f748820ed0","last_reissued_at":"2026-07-05T07:33:05.554174Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:33:05.554174Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On Biased Compression for Distributed Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.DC","math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Aleksandr Beznosikov, Mher Safaryan, Peter Richt\\'arik, Samuel Horv\\'ath","submitted_at":"2020-02-27T19:52:24Z","abstract_excerpt":"In the last few years, various communication compression techniques have emerged as an indispensable tool helping to alleviate the communication bottleneck in distributed learning. However, despite the fact biased compressors often show superior performance in practice when compared to the much more studied and understood unbiased compressors, very little is known about them. In this work we study three classes of biased compression operators, two of which are new, and their performance when applied to (stochastic) gradient descent and distributed (stochastic) gradient descent. We show for the"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2002.12410","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2002.12410/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2002.12410","created_at":"2026-07-05T07:33:05.554226+00:00"},{"alias_kind":"arxiv_version","alias_value":"2002.12410v4","created_at":"2026-07-05T07:33:05.554226+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2002.12410","created_at":"2026-07-05T07:33:05.554226+00:00"},{"alias_kind":"pith_short_12","alias_value":"5TYV2MO3O2GR","created_at":"2026-07-05T07:33:05.554226+00:00"},{"alias_kind":"pith_short_16","alias_value":"5TYV2MO3O2GRJVMW","created_at":"2026-07-05T07:33:05.554226+00:00"},{"alias_kind":"pith_short_8","alias_value":"5TYV2MO3","created_at":"2026-07-05T07:33:05.554226+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2410.15368","citing_title":"Tighter Performance Theory of FedExProx","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18174","citing_title":"Ringmaster LMO: Asynchronous Linear Minimization Oracle Momentum Method","ref_index":103,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13434","citing_title":"Rescaled Asynchronous SGD: Optimal Distributed Optimization under Data and System Heterogeneity","ref_index":218,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08871","citing_title":"Rennala MVR: Improved Time Complexity for Parallel Stochastic Optimization via Momentum-Based Variance Reduction","ref_index":101,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08236","citing_title":"Improved Convergence for Decentralized Stochastic Optimization with Biased Gradients","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07795","citing_title":"Scalable Distributed Stochastic Optimization via Bidirectional Compression: Beyond Pessimistic Limits","ref_index":25,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5TYV2MO3O2GRJVMWVDAE4BBAWI","json":"https://pith.science/pith/5TYV2MO3O2GRJVMWVDAE4BBAWI.json","graph_json":"https://pith.science/api/pith-number/5TYV2MO3O2GRJVMWVDAE4BBAWI/graph.json","events_json":"https://pith.science/api/pith-number/5TYV2MO3O2GRJVMWVDAE4BBAWI/events.json","paper":"https://pith.science/paper/5TYV2MO3"},"agent_actions":{"view_html":"https://pith.science/pith/5TYV2MO3O2GRJVMWVDAE4BBAWI","download_json":"https://pith.science/pith/5TYV2MO3O2GRJVMWVDAE4BBAWI.json","view_paper":"https://pith.science/paper/5TYV2MO3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2002.12410&json=true","fetch_graph":"https://pith.science/api/pith-number/5TYV2MO3O2GRJVMWVDAE4BBAWI/graph.json","fetch_events":"https://pith.science/api/pith-number/5TYV2MO3O2GRJVMWVDAE4BBAWI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5TYV2MO3O2GRJVMWVDAE4BBAWI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5TYV2MO3O2GRJVMWVDAE4BBAWI/action/storage_attestation","attest_author":"https://pith.science/pith/5TYV2MO3O2GRJVMWVDAE4BBAWI/action/author_attestation","sign_citation":"https://pith.science/pith/5TYV2MO3O2GRJVMWVDAE4BBAWI/action/citation_signature","submit_replication":"https://pith.science/pith/5TYV2MO3O2GRJVMWVDAE4BBAWI/action/replication_record"}},"created_at":"2026-07-05T07:33:05.554226+00:00","updated_at":"2026-07-05T07:33:05.554226+00:00"}