{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:VVNFYRGLMVJW23YUTWNIZQTWAL","short_pith_number":"pith:VVNFYRGL","schema_version":"1.0","canonical_sha256":"ad5a5c44cb65536d6f149d9a8cc27602ca1d7025d86dcd0f2c3c5d4a466c4158","source":{"kind":"arxiv","id":"2405.12755","version":2},"attestation_state":"computed","paper":{"title":"Progress Measures for Grokking on Real-world Tasks","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Satvik Golechha","submitted_at":"2024-05-21T13:06:41Z","abstract_excerpt":"Grokking, a phenomenon where machine learning models generalize long after overfitting, has been primarily observed and studied in algorithmic tasks. This paper explores grokking in real-world datasets using deep neural networks for classification under the cross-entropy loss. We challenge the prevalent hypothesis that the $L_2$ norm of weights is the primary cause of grokking by demonstrating that grokking can occur outside the expected range of weight norms. To better understand grokking, we introduce three new progress measures: activation sparsity, absolute weight entropy, and approximate "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.12755","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-21T13:06:41Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"6fcb93a3adfc7b5055633c4f3da15602f2209b50ae1198048e0bf3a7d2d2efb4","abstract_canon_sha256":"4dc0d9728ae1f4322c13555cc9913672f637419c4269401f674784e3fedb165b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:34:19.236882Z","signature_b64":"wx/u4Fvy7dZkji5NmtmL6lCze5fynEK+f1yb3Al2hTfMaoY2sP0V46EZv4mt+5eNRoj2xYHK60LuOZRjzrfhBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ad5a5c44cb65536d6f149d9a8cc27602ca1d7025d86dcd0f2c3c5d4a466c4158","last_reissued_at":"2026-07-05T08:34:19.236429Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:34:19.236429Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Progress Measures for Grokking on Real-world Tasks","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Satvik Golechha","submitted_at":"2024-05-21T13:06:41Z","abstract_excerpt":"Grokking, a phenomenon where machine learning models generalize long after overfitting, has been primarily observed and studied in algorithmic tasks. This paper explores grokking in real-world datasets using deep neural networks for classification under the cross-entropy loss. We challenge the prevalent hypothesis that the $L_2$ norm of weights is the primary cause of grokking by demonstrating that grokking can occur outside the expected range of weight norms. To better understand grokking, we introduce three new progress measures: activation sparsity, absolute weight entropy, and approximate "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.12755","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.12755/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.12755","created_at":"2026-07-05T08:34:19.236484+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.12755v2","created_at":"2026-07-05T08:34:19.236484+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.12755","created_at":"2026-07-05T08:34:19.236484+00:00"},{"alias_kind":"pith_short_12","alias_value":"VVNFYRGLMVJW","created_at":"2026-07-05T08:34:19.236484+00:00"},{"alias_kind":"pith_short_16","alias_value":"VVNFYRGLMVJW23YU","created_at":"2026-07-05T08:34:19.236484+00:00"},{"alias_kind":"pith_short_8","alias_value":"VVNFYRGL","created_at":"2026-07-05T08:34:19.236484+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.06152","citing_title":"Grokking or Glitching? How Low-Precision Drives Slingshot Loss Spikes","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VVNFYRGLMVJW23YUTWNIZQTWAL","json":"https://pith.science/pith/VVNFYRGLMVJW23YUTWNIZQTWAL.json","graph_json":"https://pith.science/api/pith-number/VVNFYRGLMVJW23YUTWNIZQTWAL/graph.json","events_json":"https://pith.science/api/pith-number/VVNFYRGLMVJW23YUTWNIZQTWAL/events.json","paper":"https://pith.science/paper/VVNFYRGL"},"agent_actions":{"view_html":"https://pith.science/pith/VVNFYRGLMVJW23YUTWNIZQTWAL","download_json":"https://pith.science/pith/VVNFYRGLMVJW23YUTWNIZQTWAL.json","view_paper":"https://pith.science/paper/VVNFYRGL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.12755&json=true","fetch_graph":"https://pith.science/api/pith-number/VVNFYRGLMVJW23YUTWNIZQTWAL/graph.json","fetch_events":"https://pith.science/api/pith-number/VVNFYRGLMVJW23YUTWNIZQTWAL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VVNFYRGLMVJW23YUTWNIZQTWAL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VVNFYRGLMVJW23YUTWNIZQTWAL/action/storage_attestation","attest_author":"https://pith.science/pith/VVNFYRGLMVJW23YUTWNIZQTWAL/action/author_attestation","sign_citation":"https://pith.science/pith/VVNFYRGLMVJW23YUTWNIZQTWAL/action/citation_signature","submit_replication":"https://pith.science/pith/VVNFYRGLMVJW23YUTWNIZQTWAL/action/replication_record"}},"created_at":"2026-07-05T08:34:19.236484+00:00","updated_at":"2026-07-05T08:34:19.236484+00:00"}