{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:LJCFK5B2WWNBI3F5LQVZ7VRZYJ","short_pith_number":"pith:LJCFK5B2","schema_version":"1.0","canonical_sha256":"5a4455743ab59a146cbd5c2b9fd639c25ef9723d9db75c8547885ee1417f348b","source":{"kind":"arxiv","id":"2301.11235","version":3},"attestation_state":"computed","paper":{"title":"Handbook of Convergence Theorems for (Stochastic) Gradient Methods","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"math.OC","authors_text":"Guillaume Garrigos, Robert M. Gower","submitted_at":"2023-01-26T17:18:36Z","abstract_excerpt":"This is a handbook of simple proofs of the convergence of gradient and stochastic gradient descent type methods. We consider functions that are Lipschitz, smooth, convex, strongly convex, and/or Polyak-{\\L}ojasiewicz functions. Our focus is on ``good proofs'' that are also simple. Each section can be consulted separately. We start with proofs of gradient descent, then on stochastic variants, including minibatching and momentum. Then move on to nonsmooth problems with the subgradient method, the proximal gradient descent and their stochastic variants. Our focus is on global convergence rates an"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2301.11235","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"math.OC","submitted_at":"2023-01-26T17:18:36Z","cross_cats_sorted":[],"title_canon_sha256":"6add0f686bb087d8dafc7a67219b11d43c19a858b202ada8f8d00854dc7c255a","abstract_canon_sha256":"d7cb0d476c46e1094b7193cddb63a510bcb65647e1a8c2ea42224f9143eaa876"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:53:59.207392Z","signature_b64":"ySgC2r6yFzrLSw91qXItLy+eWIqBmSlW0sriIGUdqsEyyaSmzLe60vofhvYU0lC1SGBEPp78gg9pK4zspflSAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5a4455743ab59a146cbd5c2b9fd639c25ef9723d9db75c8547885ee1417f348b","last_reissued_at":"2026-07-05T07:53:59.206906Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:53:59.206906Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Handbook of Convergence Theorems for (Stochastic) Gradient Methods","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"math.OC","authors_text":"Guillaume Garrigos, Robert M. Gower","submitted_at":"2023-01-26T17:18:36Z","abstract_excerpt":"This is a handbook of simple proofs of the convergence of gradient and stochastic gradient descent type methods. We consider functions that are Lipschitz, smooth, convex, strongly convex, and/or Polyak-{\\L}ojasiewicz functions. Our focus is on ``good proofs'' that are also simple. Each section can be consulted separately. We start with proofs of gradient descent, then on stochastic variants, including minibatching and momentum. Then move on to nonsmooth problems with the subgradient method, the proximal gradient descent and their stochastic variants. Our focus is on global convergence rates an"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2301.11235","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2301.11235/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2301.11235","created_at":"2026-07-05T07:53:59.206964+00:00"},{"alias_kind":"arxiv_version","alias_value":"2301.11235v3","created_at":"2026-07-05T07:53:59.206964+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2301.11235","created_at":"2026-07-05T07:53:59.206964+00:00"},{"alias_kind":"pith_short_12","alias_value":"LJCFK5B2WWNB","created_at":"2026-07-05T07:53:59.206964+00:00"},{"alias_kind":"pith_short_16","alias_value":"LJCFK5B2WWNBI3F5","created_at":"2026-07-05T07:53:59.206964+00:00"},{"alias_kind":"pith_short_8","alias_value":"LJCFK5B2","created_at":"2026-07-05T07:53:59.206964+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":29,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.23017","citing_title":"Complex Stochastic Gradient Descent and Directional Bias in Reproducing Kernel Hilbert Spaces","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27171","citing_title":"Stochastic Gradient Optimization with Model-Assisted Sampling","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00665","citing_title":"Effective dynamics of the Sinkhorn algorithm in the regime of low entropy regularization","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02365","citing_title":"FOAM: Frequency and Operator Error-Based Adaptive Damping Method for Reducing Staleness-Oriented Error for Shampoo","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14663","citing_title":"Optimal Asymptotic Rates for (Stochastic) Gradient Descent under the Local PL-Condition: A Geometric Approach","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.32005","citing_title":"Random Reshuffling Dominates Stochastic Gradient Descent","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28973","citing_title":"Sharp $O(1/k)$ convergence rate for the Sinkhorn algorithm via a local analysis","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29593","citing_title":"How AI settled the complexity of the oldest SGD algorithm","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30310","citing_title":"Highly Data Parallelizable Estimation of the Sliced-Wasserstein Distance Using Cumulative Distribution Functions","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25034","citing_title":"Randomized conjugate gradient least squares","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25499","citing_title":"Accelerated Dynamic Importance Weighting with Versatile Divergence-Minimizing Estimators","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29304","citing_title":"On subspace-constrained preconditioning for randomized iterative methods","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2603.10067","citing_title":"HTMuon: Improving Muon via Heavy-Tailed Spectral Correction","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2602.18718","citing_title":"Stochastic Gradient Variational Inference with Price's Gradient Estimator from Bures-Wasserstein to Parameter Space","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16017","citing_title":"Accelerated Gradient Descent for Faster Convergence with Minimal Overhead","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18675","citing_title":"COOPO: Cyclic Offline-Online Policy Optimization Algorithm","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19291","citing_title":"Factor Augmented High-Dimensional SGD","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2505.23737","citing_title":"On the Convergence Analysis of Muon","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2508.09103","citing_title":"Constrained free energy minimization for the design of thermal states and stabilizer thermodynamic systems","ref_index":77,"is_internal_anchor":false},{"citing_arxiv_id":"2509.02912","citing_title":"Stochastic versus Deterministic in Stochastic Gradient Descent","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2510.04686","citing_title":"How does the optimizer implicitly bias the model merging loss landscape?","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2510.07922","citing_title":"SketchGuard: Scaling Byzantine-Robust Decentralized Federated Learning via Sketch-Based Screening","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2512.18248","citing_title":"On the Convergence Rate of LoRA Gradient Descent","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22581","citing_title":"Stochastic Krasnoselskii-Mann Iterations: Convergence without Uniformly Bounded Variance","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25613","citing_title":"One Coordinate at a Time: Convergence Guarantees for Rotosolve in Variational Quantum Algorithms","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LJCFK5B2WWNBI3F5LQVZ7VRZYJ","json":"https://pith.science/pith/LJCFK5B2WWNBI3F5LQVZ7VRZYJ.json","graph_json":"https://pith.science/api/pith-number/LJCFK5B2WWNBI3F5LQVZ7VRZYJ/graph.json","events_json":"https://pith.science/api/pith-number/LJCFK5B2WWNBI3F5LQVZ7VRZYJ/events.json","paper":"https://pith.science/paper/LJCFK5B2"},"agent_actions":{"view_html":"https://pith.science/pith/LJCFK5B2WWNBI3F5LQVZ7VRZYJ","download_json":"https://pith.science/pith/LJCFK5B2WWNBI3F5LQVZ7VRZYJ.json","view_paper":"https://pith.science/paper/LJCFK5B2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2301.11235&json=true","fetch_graph":"https://pith.science/api/pith-number/LJCFK5B2WWNBI3F5LQVZ7VRZYJ/graph.json","fetch_events":"https://pith.science/api/pith-number/LJCFK5B2WWNBI3F5LQVZ7VRZYJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LJCFK5B2WWNBI3F5LQVZ7VRZYJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LJCFK5B2WWNBI3F5LQVZ7VRZYJ/action/storage_attestation","attest_author":"https://pith.science/pith/LJCFK5B2WWNBI3F5LQVZ7VRZYJ/action/author_attestation","sign_citation":"https://pith.science/pith/LJCFK5B2WWNBI3F5LQVZ7VRZYJ/action/citation_signature","submit_replication":"https://pith.science/pith/LJCFK5B2WWNBI3F5LQVZ7VRZYJ/action/replication_record"}},"created_at":"2026-07-05T07:53:59.206964+00:00","updated_at":"2026-07-05T07:53:59.206964+00:00"}