{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:HNL4SXOSQEFIOUXMCEFOHL6WZM","short_pith_number":"pith:HNL4SXOS","schema_version":"1.0","canonical_sha256":"3b57c95dd2810a8752ec110ae3afd6cb3911487a12883de69b8751d0d23d8a24","source":{"kind":"arxiv","id":"2310.16441","version":1},"attestation_state":"computed","paper":{"title":"Grokking in Linear Estimators -- A Solvable Model that Groks without Understanding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cond-mat.dis-nn","cs.LG","math-ph","math.MP"],"primary_cat":"stat.ML","authors_text":"Alon Beck, Noam Levi, Yohai Bar-Sinai","submitted_at":"2023-10-25T08:08:44Z","abstract_excerpt":"Grokking is the intriguing phenomenon where a model learns to generalize long after it has fit the training data. We show both analytically and numerically that grokking can surprisingly occur in linear networks performing linear tasks in a simple teacher-student setup with Gaussian inputs. In this setting, the full training dynamics is derived in terms of the training and generalization data covariance matrix. We present exact predictions on how the grokking time depends on input and output dimensionality, train sample size, regularization, and network initialization. We demonstrate that the "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.16441","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"stat.ML","submitted_at":"2023-10-25T08:08:44Z","cross_cats_sorted":["cond-mat.dis-nn","cs.LG","math-ph","math.MP"],"title_canon_sha256":"9889d638e0682d309227d50ff93e61c7504f50feb83c548c315070c4fa4f063d","abstract_canon_sha256":"faa62c54dedd59d9528eaca1d33a86149319046b926d75da7027c8f68a874361"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:40:58.148820Z","signature_b64":"VfxnIoQiGYllwh7lILO1eS9fEszQftjXBgKUa5065Thsj25JcxhpTXonDfD9qeygdAxfuEYpMUocgMNuueFzBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3b57c95dd2810a8752ec110ae3afd6cb3911487a12883de69b8751d0d23d8a24","last_reissued_at":"2026-07-05T07:40:58.148396Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:40:58.148396Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Grokking in Linear Estimators -- A Solvable Model that Groks without Understanding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cond-mat.dis-nn","cs.LG","math-ph","math.MP"],"primary_cat":"stat.ML","authors_text":"Alon Beck, Noam Levi, Yohai Bar-Sinai","submitted_at":"2023-10-25T08:08:44Z","abstract_excerpt":"Grokking is the intriguing phenomenon where a model learns to generalize long after it has fit the training data. We show both analytically and numerically that grokking can surprisingly occur in linear networks performing linear tasks in a simple teacher-student setup with Gaussian inputs. In this setting, the full training dynamics is derived in terms of the training and generalization data covariance matrix. We present exact predictions on how the grokking time depends on input and output dimensionality, train sample size, regularization, and network initialization. We demonstrate that the "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.16441","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.16441/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.16441","created_at":"2026-07-05T07:40:58.148452+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.16441v1","created_at":"2026-07-05T07:40:58.148452+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.16441","created_at":"2026-07-05T07:40:58.148452+00:00"},{"alias_kind":"pith_short_12","alias_value":"HNL4SXOSQEFI","created_at":"2026-07-05T07:40:58.148452+00:00"},{"alias_kind":"pith_short_16","alias_value":"HNL4SXOSQEFIOUXM","created_at":"2026-07-05T07:40:58.148452+00:00"},{"alias_kind":"pith_short_8","alias_value":"HNL4SXOS","created_at":"2026-07-05T07:40:58.148452+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20299","citing_title":"Statistical Properties of Training & Generalization","ref_index":123,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17120","citing_title":"Noise-Driven Escape from Metastable Phases explains Grokking in Deep Neural Networks","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20299","citing_title":"Statistical Properties of Training & Generalization","ref_index":123,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17767","citing_title":"Feature Learning in Linear-Width Two-Layer Networks: Two vs. One Step of Gradient Descent","ref_index":206,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17767","citing_title":"Feature Learning in Linear-Width Two-Layer Networks: Two vs. One Step of Gradient Descent","ref_index":206,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HNL4SXOSQEFIOUXMCEFOHL6WZM","json":"https://pith.science/pith/HNL4SXOSQEFIOUXMCEFOHL6WZM.json","graph_json":"https://pith.science/api/pith-number/HNL4SXOSQEFIOUXMCEFOHL6WZM/graph.json","events_json":"https://pith.science/api/pith-number/HNL4SXOSQEFIOUXMCEFOHL6WZM/events.json","paper":"https://pith.science/paper/HNL4SXOS"},"agent_actions":{"view_html":"https://pith.science/pith/HNL4SXOSQEFIOUXMCEFOHL6WZM","download_json":"https://pith.science/pith/HNL4SXOSQEFIOUXMCEFOHL6WZM.json","view_paper":"https://pith.science/paper/HNL4SXOS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.16441&json=true","fetch_graph":"https://pith.science/api/pith-number/HNL4SXOSQEFIOUXMCEFOHL6WZM/graph.json","fetch_events":"https://pith.science/api/pith-number/HNL4SXOSQEFIOUXMCEFOHL6WZM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HNL4SXOSQEFIOUXMCEFOHL6WZM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HNL4SXOSQEFIOUXMCEFOHL6WZM/action/storage_attestation","attest_author":"https://pith.science/pith/HNL4SXOSQEFIOUXMCEFOHL6WZM/action/author_attestation","sign_citation":"https://pith.science/pith/HNL4SXOSQEFIOUXMCEFOHL6WZM/action/citation_signature","submit_replication":"https://pith.science/pith/HNL4SXOSQEFIOUXMCEFOHL6WZM/action/replication_record"}},"created_at":"2026-07-05T07:40:58.148452+00:00","updated_at":"2026-07-05T07:40:58.148452+00:00"}