{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:RHDJK25A7NRYKQQC57Q35JB4UZ","short_pith_number":"pith:RHDJK25A","schema_version":"1.0","canonical_sha256":"89c6956ba0fb63854202efe1bea43ca672380ed53d907e705621b42a22c72a21","source":{"kind":"arxiv","id":"2501.04697","version":2},"attestation_state":"computed","paper":{"title":"Grokking at the Edge of Numerical Stability","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV","stat.ML"],"primary_cat":"cs.LG","authors_text":"Lucas Prieto, Melih Barsbey, Pedro A.M. Mediano, Tolga Birdal","submitted_at":"2025-01-08T18:58:48Z","abstract_excerpt":"Grokking, the sudden generalization that occurs after prolonged overfitting, is a surprising phenomenon challenging our understanding of deep learning. Although significant progress has been made in understanding grokking, the reasons behind the delayed generalization and its dependence on regularization remain unclear. In this work, we argue that without regularization, grokking tasks push models to the edge of numerical stability, introducing floating point errors in the Softmax function, which we refer to as Softmax Collapse (SC). We demonstrate that SC prevents grokking and that mitigating"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.04697","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-01-08T18:58:48Z","cross_cats_sorted":["cs.AI","cs.CV","stat.ML"],"title_canon_sha256":"215cdfb109b560a08ac4d1419136dd6d2c0202d5484f440de0c4124549eebcf8","abstract_canon_sha256":"d280086f7ecbf98b7f830333543a8f8e8516ff8b5f505cfdef45bb978d4372ac"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:04:43.733850Z","signature_b64":"WCdBXkTBWn2oZmrw7edZ4HLRZBtkTihMTkeykmYW31XWRnOAwb6qHsAA+v7mTDnSthpKJDcaB9Jo31YgpUKEBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"89c6956ba0fb63854202efe1bea43ca672380ed53d907e705621b42a22c72a21","last_reissued_at":"2026-07-05T11:04:43.733306Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:04:43.733306Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Grokking at the Edge of Numerical Stability","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV","stat.ML"],"primary_cat":"cs.LG","authors_text":"Lucas Prieto, Melih Barsbey, Pedro A.M. Mediano, Tolga Birdal","submitted_at":"2025-01-08T18:58:48Z","abstract_excerpt":"Grokking, the sudden generalization that occurs after prolonged overfitting, is a surprising phenomenon challenging our understanding of deep learning. Although significant progress has been made in understanding grokking, the reasons behind the delayed generalization and its dependence on regularization remain unclear. In this work, we argue that without regularization, grokking tasks push models to the edge of numerical stability, introducing floating point errors in the Softmax function, which we refer to as Softmax Collapse (SC). We demonstrate that SC prevents grokking and that mitigating"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.04697","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.04697/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.04697","created_at":"2026-07-05T11:04:43.733378+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.04697v2","created_at":"2026-07-05T11:04:43.733378+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.04697","created_at":"2026-07-05T11:04:43.733378+00:00"},{"alias_kind":"pith_short_12","alias_value":"RHDJK25A7NRY","created_at":"2026-07-05T11:04:43.733378+00:00"},{"alias_kind":"pith_short_16","alias_value":"RHDJK25A7NRYKQQC","created_at":"2026-07-05T11:04:43.733378+00:00"},{"alias_kind":"pith_short_8","alias_value":"RHDJK25A","created_at":"2026-07-05T11:04:43.733378+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18465","citing_title":"What Does the Weight Norm Control in Grokking? Logit-Scale Mediation under Cross-Entropy","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29176","citing_title":"Dead-Direction Conditioners: Gauge-Equivariant Preconditioning for Deep Networks","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2510.04930","citing_title":"Egalitarian Gradient Descent: A Simple Approach to Accelerated Grokking","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2504.20571","citing_title":"Reinforcement Learning for Reasoning in Large Language Models with One Training Example","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2510.04871","citing_title":"Less is More: Recursive Reasoning with Tiny Networks","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08237","citing_title":"Distributional Spectral Diagnostics for Localizing Grokking Transitions","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RHDJK25A7NRYKQQC57Q35JB4UZ","json":"https://pith.science/pith/RHDJK25A7NRYKQQC57Q35JB4UZ.json","graph_json":"https://pith.science/api/pith-number/RHDJK25A7NRYKQQC57Q35JB4UZ/graph.json","events_json":"https://pith.science/api/pith-number/RHDJK25A7NRYKQQC57Q35JB4UZ/events.json","paper":"https://pith.science/paper/RHDJK25A"},"agent_actions":{"view_html":"https://pith.science/pith/RHDJK25A7NRYKQQC57Q35JB4UZ","download_json":"https://pith.science/pith/RHDJK25A7NRYKQQC57Q35JB4UZ.json","view_paper":"https://pith.science/paper/RHDJK25A","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.04697&json=true","fetch_graph":"https://pith.science/api/pith-number/RHDJK25A7NRYKQQC57Q35JB4UZ/graph.json","fetch_events":"https://pith.science/api/pith-number/RHDJK25A7NRYKQQC57Q35JB4UZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RHDJK25A7NRYKQQC57Q35JB4UZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RHDJK25A7NRYKQQC57Q35JB4UZ/action/storage_attestation","attest_author":"https://pith.science/pith/RHDJK25A7NRYKQQC57Q35JB4UZ/action/author_attestation","sign_citation":"https://pith.science/pith/RHDJK25A7NRYKQQC57Q35JB4UZ/action/citation_signature","submit_replication":"https://pith.science/pith/RHDJK25A7NRYKQQC57Q35JB4UZ/action/replication_record"}},"created_at":"2026-07-05T11:04:43.733378+00:00","updated_at":"2026-07-05T11:04:43.733378+00:00"}