{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:5EASHK6R2T5MAIF3ES27BFRGCZ","short_pith_number":"pith:5EASHK6R","schema_version":"1.0","canonical_sha256":"e90123abd1d4fac020bb24b5f09626167922f632d76e005d55b7888829ea7a80","source":{"kind":"arxiv","id":"2312.16143","version":2},"attestation_state":"computed","paper":{"title":"On the Trajectories of SGD Without Replacement","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Pierfrancesco Beneventano","submitted_at":"2023-12-26T18:06:48Z","abstract_excerpt":"This article examines the implicit regularization effect of Stochastic Gradient Descent (SGD). We consider the case of SGD without replacement, the variant typically used to optimize large-scale neural networks. We analyze this algorithm in a more realistic regime than typically considered in theoretical works on SGD, as, e.g., we allow the product of the learning rate and Hessian to be $O(1)$ and we do not specify any model architecture, learning task, or loss (objective) function. Our core theoretical result is that optimizing with SGD without replacement is locally equivalent to making an a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.16143","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-12-26T18:06:48Z","cross_cats_sorted":["math.OC","stat.ML"],"title_canon_sha256":"9b021bc6e9342feaaf64d5f9e48dca776541d88b171c222285de2863ca4a12ec","abstract_canon_sha256":"d7a37e0ca38b922be8dfe1fd36adb15a1e06c09770ceebd7171a114c36da46c5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:10:21.336483Z","signature_b64":"cB/U2VfrKWrnfOP7W4Bfaut4nterS8SskDfAwsiM+2Kdqn7epw3YJgCVDCQ5FGDAkHrAOh+14X9UtWwG6ZX7DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e90123abd1d4fac020bb24b5f09626167922f632d76e005d55b7888829ea7a80","last_reissued_at":"2026-07-05T08:10:21.335965Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:10:21.335965Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On the Trajectories of SGD Without Replacement","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Pierfrancesco Beneventano","submitted_at":"2023-12-26T18:06:48Z","abstract_excerpt":"This article examines the implicit regularization effect of Stochastic Gradient Descent (SGD). We consider the case of SGD without replacement, the variant typically used to optimize large-scale neural networks. We analyze this algorithm in a more realistic regime than typically considered in theoretical works on SGD, as, e.g., we allow the product of the learning rate and Hessian to be $O(1)$ and we do not specify any model architecture, learning task, or loss (objective) function. Our core theoretical result is that optimizing with SGD without replacement is locally equivalent to making an a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.16143","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.16143/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.16143","created_at":"2026-07-05T08:10:21.336025+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.16143v2","created_at":"2026-07-05T08:10:21.336025+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.16143","created_at":"2026-07-05T08:10:21.336025+00:00"},{"alias_kind":"pith_short_12","alias_value":"5EASHK6R2T5M","created_at":"2026-07-05T08:10:21.336025+00:00"},{"alias_kind":"pith_short_16","alias_value":"5EASHK6R2T5MAIF3","created_at":"2026-07-05T08:10:21.336025+00:00"},{"alias_kind":"pith_short_8","alias_value":"5EASHK6R","created_at":"2026-07-05T08:10:21.336025+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.29152","citing_title":"Do Deep Networks Forget Initialization? A Forgetting-Time View of Practical Inductive Bias","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2602.01642","citing_title":"The Effect of Mini-Batch Noise on the Implicit Bias of Adam","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14345","citing_title":"Convergence of difference inclusions via a diameter criterion","ref_index":90,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5EASHK6R2T5MAIF3ES27BFRGCZ","json":"https://pith.science/pith/5EASHK6R2T5MAIF3ES27BFRGCZ.json","graph_json":"https://pith.science/api/pith-number/5EASHK6R2T5MAIF3ES27BFRGCZ/graph.json","events_json":"https://pith.science/api/pith-number/5EASHK6R2T5MAIF3ES27BFRGCZ/events.json","paper":"https://pith.science/paper/5EASHK6R"},"agent_actions":{"view_html":"https://pith.science/pith/5EASHK6R2T5MAIF3ES27BFRGCZ","download_json":"https://pith.science/pith/5EASHK6R2T5MAIF3ES27BFRGCZ.json","view_paper":"https://pith.science/paper/5EASHK6R","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.16143&json=true","fetch_graph":"https://pith.science/api/pith-number/5EASHK6R2T5MAIF3ES27BFRGCZ/graph.json","fetch_events":"https://pith.science/api/pith-number/5EASHK6R2T5MAIF3ES27BFRGCZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5EASHK6R2T5MAIF3ES27BFRGCZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5EASHK6R2T5MAIF3ES27BFRGCZ/action/storage_attestation","attest_author":"https://pith.science/pith/5EASHK6R2T5MAIF3ES27BFRGCZ/action/author_attestation","sign_citation":"https://pith.science/pith/5EASHK6R2T5MAIF3ES27BFRGCZ/action/citation_signature","submit_replication":"https://pith.science/pith/5EASHK6R2T5MAIF3ES27BFRGCZ/action/replication_record"}},"created_at":"2026-07-05T08:10:21.336025+00:00","updated_at":"2026-07-05T08:10:21.336025+00:00"}