{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:ABLLWIY6SXU7F5JNRKM4RR2O2M","short_pith_number":"pith:ABLLWIY6","schema_version":"1.0","canonical_sha256":"0056bb231e95e9f2f52d8a99c8c74ed30f17666a2c036fed325553446ef77573","source":{"kind":"arxiv","id":"2306.12498","version":2},"attestation_state":"computed","paper":{"title":"Empirical Risk Minimization with Shuffled SGD: A Primal-Dual Perspective and Improved Bounds","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"math.OC","authors_text":"Cheuk Yin Lin, Jelena Diakonikolas, Xufeng Cai","submitted_at":"2023-06-21T18:14:44Z","abstract_excerpt":"Stochastic gradient descent (SGD) is perhaps the most prevalent optimization method in modern machine learning. Contrary to the empirical practice of sampling from the datasets without replacement and with (possible) reshuffling at each epoch, the theoretical counterpart of SGD usually relies on the assumption of sampling with replacement. It is only very recently that SGD with sampling without replacement -- shuffled SGD -- has been analyzed. For convex finite sum problems with $n$ components and under the $L$-smoothness assumption for each component function, there are matching upper and low"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.12498","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"math.OC","submitted_at":"2023-06-21T18:14:44Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"dcb4cb5e1fe73e57f4186e3a95613477ba2c9f280b21e52b7b4408a64587df44","abstract_canon_sha256":"567484799e1e8c5bba40f6d6db100f3de58a616eee154eb354444e4b5cc49a8d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:42:17.822377Z","signature_b64":"JjHZ4cng0djrda/AVgX9hcbiC89Z9VTnjhB13Cz9BfTf+ELYv7k6B7N/pY56IUcTjSNgzbn1d6Qxf0ZqLG5jCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0056bb231e95e9f2f52d8a99c8c74ed30f17666a2c036fed325553446ef77573","last_reissued_at":"2026-07-05T07:42:17.821523Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:42:17.821523Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Empirical Risk Minimization with Shuffled SGD: A Primal-Dual Perspective and Improved Bounds","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"math.OC","authors_text":"Cheuk Yin Lin, Jelena Diakonikolas, Xufeng Cai","submitted_at":"2023-06-21T18:14:44Z","abstract_excerpt":"Stochastic gradient descent (SGD) is perhaps the most prevalent optimization method in modern machine learning. Contrary to the empirical practice of sampling from the datasets without replacement and with (possible) reshuffling at each epoch, the theoretical counterpart of SGD usually relies on the assumption of sampling with replacement. It is only very recently that SGD with sampling without replacement -- shuffled SGD -- has been analyzed. For convex finite sum problems with $n$ components and under the $L$-smoothness assumption for each component function, there are matching upper and low"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.12498","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.12498/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.12498","created_at":"2026-07-05T07:42:17.821967+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.12498v2","created_at":"2026-07-05T07:42:17.821967+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.12498","created_at":"2026-07-05T07:42:17.821967+00:00"},{"alias_kind":"pith_short_12","alias_value":"ABLLWIY6SXU7","created_at":"2026-07-05T07:42:17.821967+00:00"},{"alias_kind":"pith_short_16","alias_value":"ABLLWIY6SXU7F5JN","created_at":"2026-07-05T07:42:17.821967+00:00"},{"alias_kind":"pith_short_8","alias_value":"ABLLWIY6","created_at":"2026-07-05T07:42:17.821967+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11431","citing_title":"Mirror Descent Beyond Euclidean Stability: An Exponential Separation in Initialization Sensitivity","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10110","citing_title":"The Dual Averaging Power-Prox Method with Application to Heavy-Tail Incremental Gradient","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30930","citing_title":"SGD at the Edge of Stability: Stochastic Stabilization with Large Learning Rates","ref_index":94,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30559","citing_title":"Convergence of Continual Learning in Homogeneous Deep Networks","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10373","citing_title":"Shuffling the Data, Stretching the Step-size: Sharper Bias in constant step-size SGD","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ABLLWIY6SXU7F5JNRKM4RR2O2M","json":"https://pith.science/pith/ABLLWIY6SXU7F5JNRKM4RR2O2M.json","graph_json":"https://pith.science/api/pith-number/ABLLWIY6SXU7F5JNRKM4RR2O2M/graph.json","events_json":"https://pith.science/api/pith-number/ABLLWIY6SXU7F5JNRKM4RR2O2M/events.json","paper":"https://pith.science/paper/ABLLWIY6"},"agent_actions":{"view_html":"https://pith.science/pith/ABLLWIY6SXU7F5JNRKM4RR2O2M","download_json":"https://pith.science/pith/ABLLWIY6SXU7F5JNRKM4RR2O2M.json","view_paper":"https://pith.science/paper/ABLLWIY6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.12498&json=true","fetch_graph":"https://pith.science/api/pith-number/ABLLWIY6SXU7F5JNRKM4RR2O2M/graph.json","fetch_events":"https://pith.science/api/pith-number/ABLLWIY6SXU7F5JNRKM4RR2O2M/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ABLLWIY6SXU7F5JNRKM4RR2O2M/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ABLLWIY6SXU7F5JNRKM4RR2O2M/action/storage_attestation","attest_author":"https://pith.science/pith/ABLLWIY6SXU7F5JNRKM4RR2O2M/action/author_attestation","sign_citation":"https://pith.science/pith/ABLLWIY6SXU7F5JNRKM4RR2O2M/action/citation_signature","submit_replication":"https://pith.science/pith/ABLLWIY6SXU7F5JNRKM4RR2O2M/action/replication_record"}},"created_at":"2026-07-05T07:42:17.821967+00:00","updated_at":"2026-07-05T07:42:17.821967+00:00"}