{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:VGO5SVEVV6R4QHV4Q4WFNOHC6I","short_pith_number":"pith:VGO5SVEV","schema_version":"1.0","canonical_sha256":"a99dd95495afa3c81ebc872c56b8e2f22767f91295dcb53650fc4cae0106cb6a","source":{"kind":"arxiv","id":"2008.00051","version":2},"attestation_state":"computed","paper":{"title":"On the Convergence of SGD with Biased Gradients","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Ahmad Ajalloeian, Sebastian U. Stich","submitted_at":"2020-07-31T19:37:59Z","abstract_excerpt":"We analyze the complexity of biased stochastic gradient methods (SGD), where individual updates are corrupted by deterministic, i.e. biased error terms. We derive convergence results for smooth (non-convex) functions and give improved rates under the Polyak-Lojasiewicz condition. We quantify how the magnitude of the bias impacts the attainable accuracy and the convergence rates (sometimes leading to divergence).\n  Our framework covers many applications where either only biased gradient updates are available, or preferred, over unbiased ones for performance reasons. For instance, in the domain "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2008.00051","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-07-31T19:37:59Z","cross_cats_sorted":["math.OC","stat.ML"],"title_canon_sha256":"2501f5fd3401c23b94bed2c21dfcdbce86c4bf722f7c7a0c4c9056c22a02a9c0","abstract_canon_sha256":"66ccc7896fa4b7009320286341134cd8b5b0d84606e88523399dc8e05e9ba183"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:38:29.747718Z","signature_b64":"hwS79ELm1rJGVnvNlmjZXJ5VcPM5wMTwm67YPofAQRJw3DIJYQO9xS/H/KEWbK1CgQyaU/7ZlBgIZLrTIO2/Cw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a99dd95495afa3c81ebc872c56b8e2f22767f91295dcb53650fc4cae0106cb6a","last_reissued_at":"2026-07-05T02:38:29.747261Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:38:29.747261Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On the Convergence of SGD with Biased Gradients","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Ahmad Ajalloeian, Sebastian U. Stich","submitted_at":"2020-07-31T19:37:59Z","abstract_excerpt":"We analyze the complexity of biased stochastic gradient methods (SGD), where individual updates are corrupted by deterministic, i.e. biased error terms. We derive convergence results for smooth (non-convex) functions and give improved rates under the Polyak-Lojasiewicz condition. We quantify how the magnitude of the bias impacts the attainable accuracy and the convergence rates (sometimes leading to divergence).\n  Our framework covers many applications where either only biased gradient updates are available, or preferred, over unbiased ones for performance reasons. For instance, in the domain "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2008.00051","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2008.00051/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2008.00051","created_at":"2026-07-05T02:38:29.747322+00:00"},{"alias_kind":"arxiv_version","alias_value":"2008.00051v2","created_at":"2026-07-05T02:38:29.747322+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2008.00051","created_at":"2026-07-05T02:38:29.747322+00:00"},{"alias_kind":"pith_short_12","alias_value":"VGO5SVEVV6R4","created_at":"2026-07-05T02:38:29.747322+00:00"},{"alias_kind":"pith_short_16","alias_value":"VGO5SVEVV6R4QHV4","created_at":"2026-07-05T02:38:29.747322+00:00"},{"alias_kind":"pith_short_8","alias_value":"VGO5SVEV","created_at":"2026-07-05T02:38:29.747322+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18463","citing_title":"Mixed-Precision Communication-Avoiding SGD for Generalized Linear Models on GPUs","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20606","citing_title":"Distributed Quantum Learning over Near-term Devices: Convergence Analysis and Security Design","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2408.00929","citing_title":"Verification of Machine Unlearning is Fragile","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16520","citing_title":"Global Convergence of Sampling-Based Nonconvex Optimization through Diffusion-Style Smoothing","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16690","citing_title":"UB-SMoE: Universally Balanced Sparse Mixture-of-Experts for Resource-adaptive Federated Fine-tuning of Foundation Models","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16875","citing_title":"Stochastic Optimization and Data Science","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13434","citing_title":"Rescaled Asynchronous SGD: Optimal Distributed Optimization under Data and System Heterogeneity","ref_index":217,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05710","citing_title":"On the Blessing of Pre-training in Weak-to-Strong Generalization","ref_index":154,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15416","citing_title":"StoSignSGD: Unbiased Structural Stochasticity Fixes SignSGD for Training Large Language Models","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VGO5SVEVV6R4QHV4Q4WFNOHC6I","json":"https://pith.science/pith/VGO5SVEVV6R4QHV4Q4WFNOHC6I.json","graph_json":"https://pith.science/api/pith-number/VGO5SVEVV6R4QHV4Q4WFNOHC6I/graph.json","events_json":"https://pith.science/api/pith-number/VGO5SVEVV6R4QHV4Q4WFNOHC6I/events.json","paper":"https://pith.science/paper/VGO5SVEV"},"agent_actions":{"view_html":"https://pith.science/pith/VGO5SVEVV6R4QHV4Q4WFNOHC6I","download_json":"https://pith.science/pith/VGO5SVEVV6R4QHV4Q4WFNOHC6I.json","view_paper":"https://pith.science/paper/VGO5SVEV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2008.00051&json=true","fetch_graph":"https://pith.science/api/pith-number/VGO5SVEVV6R4QHV4Q4WFNOHC6I/graph.json","fetch_events":"https://pith.science/api/pith-number/VGO5SVEVV6R4QHV4Q4WFNOHC6I/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VGO5SVEVV6R4QHV4Q4WFNOHC6I/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VGO5SVEVV6R4QHV4Q4WFNOHC6I/action/storage_attestation","attest_author":"https://pith.science/pith/VGO5SVEVV6R4QHV4Q4WFNOHC6I/action/author_attestation","sign_citation":"https://pith.science/pith/VGO5SVEVV6R4QHV4Q4WFNOHC6I/action/citation_signature","submit_replication":"https://pith.science/pith/VGO5SVEVV6R4QHV4Q4WFNOHC6I/action/replication_record"}},"created_at":"2026-07-05T02:38:29.747322+00:00","updated_at":"2026-07-05T02:38:29.747322+00:00"}