{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2017:5SKNTGLAAFPG7V3TF33DF2HSWA","short_pith_number":"pith:5SKNTGLA","schema_version":"1.0","canonical_sha256":"ec94d99960015e6fd7732ef632e8f2b0376ad761a794790a82e120201dcdea4c","source":{"kind":"arxiv","id":"1712.06559","version":3},"attestation_state":"computed","paper":{"title":"The Power of Interpolation: Understanding the Effectiveness of SGD in Modern Over-parametrized Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Mikhail Belkin, Raef Bassily, Siyuan Ma","submitted_at":"2017-12-18T18:10:39Z","abstract_excerpt":"In this paper we aim to formally explain the phenomenon of fast convergence of SGD observed in modern machine learning. The key observation is that most modern learning architectures are over-parametrized and are trained to interpolate the data by driving the empirical loss (classification and regression) close to zero. While it is still unclear why these interpolated solutions perform well on test data, we show that these regimes allow for fast convergence of SGD, comparable in number of iterations to full gradient descent.\n  For convex loss functions we obtain an exponential convergence boun"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1712.06559","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2017-12-18T18:10:39Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"a30c6b5036366e79f924ca2915ea174e45ab2b11ad959743316f8e9354ab9a0d","abstract_canon_sha256":"138352b1eda421432d5481548296f5542c8a8fc12f12932df74da3b1ae55ac74"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-18T00:13:10.648646Z","signature_b64":"M6uuEe5og+pq3d9fmdLffQktD8l486lx1Gdu5xuSkGzdhMPwSMxPlMF+ntiqBIRXlKZ31KDJ2O5fq0+sID8KDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ec94d99960015e6fd7732ef632e8f2b0376ad761a794790a82e120201dcdea4c","last_reissued_at":"2026-05-18T00:13:10.647994Z","signature_status":"signed_v1","first_computed_at":"2026-05-18T00:13:10.647994Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Power of Interpolation: Understanding the Effectiveness of SGD in Modern Over-parametrized Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Mikhail Belkin, Raef Bassily, Siyuan Ma","submitted_at":"2017-12-18T18:10:39Z","abstract_excerpt":"In this paper we aim to formally explain the phenomenon of fast convergence of SGD observed in modern machine learning. The key observation is that most modern learning architectures are over-parametrized and are trained to interpolate the data by driving the empirical loss (classification and regression) close to zero. While it is still unclear why these interpolated solutions perform well on test data, we show that these regimes allow for fast convergence of SGD, comparable in number of iterations to full gradient descent.\n  For convex loss functions we obtain an exponential convergence boun"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1712.06559","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1712.06559","created_at":"2026-05-18T00:13:10.648096+00:00"},{"alias_kind":"arxiv_version","alias_value":"1712.06559v3","created_at":"2026-05-18T00:13:10.648096+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1712.06559","created_at":"2026-05-18T00:13:10.648096+00:00"},{"alias_kind":"pith_short_12","alias_value":"5SKNTGLAAFPG","created_at":"2026-05-18T12:31:00.734936+00:00"},{"alias_kind":"pith_short_16","alias_value":"5SKNTGLAAFPG7V3T","created_at":"2026-05-18T12:31:00.734936+00:00"},{"alias_kind":"pith_short_8","alias_value":"5SKNTGLA","created_at":"2026-05-18T12:31:00.734936+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"1907.10732","citing_title":"Hessian based analysis of SGD for Deep Nets: Dynamics and Generalization","ref_index":42,"is_internal_anchor":true},{"citing_arxiv_id":"2102.01293","citing_title":"Scaling Laws for Transfer","ref_index":103,"is_internal_anchor":true},{"citing_arxiv_id":"2010.14701","citing_title":"Scaling Laws for Autoregressive Generative Modeling","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2112.00861","citing_title":"A General Language Assistant as a Laboratory for Alignment","ref_index":145,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06959","citing_title":"Locally Near Optimal Piecewise Linear Regression in High Dimensions via Difference of Max-Affine Functions","ref_index":201,"is_internal_anchor":false},{"citing_arxiv_id":"2207.05221","citing_title":"Language Models (Mostly) Know What They Know","ref_index":222,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5SKNTGLAAFPG7V3TF33DF2HSWA","json":"https://pith.science/pith/5SKNTGLAAFPG7V3TF33DF2HSWA.json","graph_json":"https://pith.science/api/pith-number/5SKNTGLAAFPG7V3TF33DF2HSWA/graph.json","events_json":"https://pith.science/api/pith-number/5SKNTGLAAFPG7V3TF33DF2HSWA/events.json","paper":"https://pith.science/paper/5SKNTGLA"},"agent_actions":{"view_html":"https://pith.science/pith/5SKNTGLAAFPG7V3TF33DF2HSWA","download_json":"https://pith.science/pith/5SKNTGLAAFPG7V3TF33DF2HSWA.json","view_paper":"https://pith.science/paper/5SKNTGLA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1712.06559&json=true","fetch_graph":"https://pith.science/api/pith-number/5SKNTGLAAFPG7V3TF33DF2HSWA/graph.json","fetch_events":"https://pith.science/api/pith-number/5SKNTGLAAFPG7V3TF33DF2HSWA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5SKNTGLAAFPG7V3TF33DF2HSWA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5SKNTGLAAFPG7V3TF33DF2HSWA/action/storage_attestation","attest_author":"https://pith.science/pith/5SKNTGLAAFPG7V3TF33DF2HSWA/action/author_attestation","sign_citation":"https://pith.science/pith/5SKNTGLAAFPG7V3TF33DF2HSWA/action/citation_signature","submit_replication":"https://pith.science/pith/5SKNTGLAAFPG7V3TF33DF2HSWA/action/replication_record"}},"created_at":"2026-05-18T00:13:10.648096+00:00","updated_at":"2026-05-18T00:13:10.648096+00:00"}