{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:H7TWAQSLGB4HLVB75TMH6YS6CF","short_pith_number":"pith:H7TWAQSL","schema_version":"1.0","canonical_sha256":"3fe760424b307875d43fecd87f625e116780d8eb5fdbf165c970c68f8d187d08","source":{"kind":"arxiv","id":"2311.06597","version":2},"attestation_state":"computed","paper":{"title":"Understanding Grokking Through A Robustness Viewpoint","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Weiran Huang, Zhiquan Tan","submitted_at":"2023-11-11T15:45:44Z","abstract_excerpt":"Recently, an interesting phenomenon called grokking has gained much attention, where generalization occurs long after the models have initially overfitted the training data. We try to understand this seemingly strange phenomenon through the robustness of the neural network. From a robustness perspective, we show that the popular $l_2$ weight norm (metric) of the neural network is actually a sufficient condition for grokking. Based on the previous observations, we propose perturbation-based methods to speed up the generalization process. In addition, we examine the standard training process on "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.06597","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-11-11T15:45:44Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"8ed424a9d0699eb1d022b681af99fc60221604f0ecd4f15fd3d348972196cfd7","abstract_canon_sha256":"c77da1163579b6c59d4dbf5ce7f39ae76c4e8c1eb0e4331ff4cc2fc56968e1e0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:40:43.008294Z","signature_b64":"SmIwfrsUFYbOJ5qWt5EZLGD8X5QPM1qSPDoV5V+LRwfx6IFe/PW/ot7FiJDfyoTSGpJjyBRYmR2sBAEM+H6iDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3fe760424b307875d43fecd87f625e116780d8eb5fdbf165c970c68f8d187d08","last_reissued_at":"2026-07-05T07:40:43.007788Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:40:43.007788Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Understanding Grokking Through A Robustness Viewpoint","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Weiran Huang, Zhiquan Tan","submitted_at":"2023-11-11T15:45:44Z","abstract_excerpt":"Recently, an interesting phenomenon called grokking has gained much attention, where generalization occurs long after the models have initially overfitted the training data. We try to understand this seemingly strange phenomenon through the robustness of the neural network. From a robustness perspective, we show that the popular $l_2$ weight norm (metric) of the neural network is actually a sufficient condition for grokking. Based on the previous observations, we propose perturbation-based methods to speed up the generalization process. In addition, we examine the standard training process on "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.06597","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.06597/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.06597","created_at":"2026-07-05T07:40:43.007859+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.06597v2","created_at":"2026-07-05T07:40:43.007859+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.06597","created_at":"2026-07-05T07:40:43.007859+00:00"},{"alias_kind":"pith_short_12","alias_value":"H7TWAQSLGB4H","created_at":"2026-07-05T07:40:43.007859+00:00"},{"alias_kind":"pith_short_16","alias_value":"H7TWAQSLGB4HLVB7","created_at":"2026-07-05T07:40:43.007859+00:00"},{"alias_kind":"pith_short_8","alias_value":"H7TWAQSL","created_at":"2026-07-05T07:40:43.007859+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.12284","citing_title":"GrokAlign: Geometric Characterisation and Acceleration of Grokking","ref_index":39,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/H7TWAQSLGB4HLVB75TMH6YS6CF","json":"https://pith.science/pith/H7TWAQSLGB4HLVB75TMH6YS6CF.json","graph_json":"https://pith.science/api/pith-number/H7TWAQSLGB4HLVB75TMH6YS6CF/graph.json","events_json":"https://pith.science/api/pith-number/H7TWAQSLGB4HLVB75TMH6YS6CF/events.json","paper":"https://pith.science/paper/H7TWAQSL"},"agent_actions":{"view_html":"https://pith.science/pith/H7TWAQSLGB4HLVB75TMH6YS6CF","download_json":"https://pith.science/pith/H7TWAQSLGB4HLVB75TMH6YS6CF.json","view_paper":"https://pith.science/paper/H7TWAQSL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.06597&json=true","fetch_graph":"https://pith.science/api/pith-number/H7TWAQSLGB4HLVB75TMH6YS6CF/graph.json","fetch_events":"https://pith.science/api/pith-number/H7TWAQSLGB4HLVB75TMH6YS6CF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/H7TWAQSLGB4HLVB75TMH6YS6CF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/H7TWAQSLGB4HLVB75TMH6YS6CF/action/storage_attestation","attest_author":"https://pith.science/pith/H7TWAQSLGB4HLVB75TMH6YS6CF/action/author_attestation","sign_citation":"https://pith.science/pith/H7TWAQSLGB4HLVB75TMH6YS6CF/action/citation_signature","submit_replication":"https://pith.science/pith/H7TWAQSLGB4HLVB75TMH6YS6CF/action/replication_record"}},"created_at":"2026-07-05T07:40:43.007859+00:00","updated_at":"2026-07-05T07:40:43.007859+00:00"}