{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:RGZO66YDJ7YSLSSIPYYXJONM4W","short_pith_number":"pith:RGZO66YD","schema_version":"1.0","canonical_sha256":"89b2ef7b034ff125ca487e3174b9ace5a7b58c723ba4dafefc7248be319822bd","source":{"kind":"arxiv","id":"2405.15071","version":3},"attestation_state":"computed","paper":{"title":"Grokked Transformers are Implicit Reasoners: A Mechanistic Journey to the Edge of Generalization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Boshi Wang, Huan Sun, Xiang Yue, Yu Su","submitted_at":"2024-05-23T21:42:19Z","abstract_excerpt":"We study whether transformers can learn to implicitly reason over parametric knowledge, a skill that even the most capable language models struggle with. Focusing on two representative reasoning types, composition and comparison, we consistently find that transformers can learn implicit reasoning, but only through grokking, i.e., extended training far beyond overfitting. The levels of generalization also vary across reasoning types: when faced with out-of-distribution examples, transformers fail to systematically generalize for composition but succeed for comparison. We delve into the model's "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.15071","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-05-23T21:42:19Z","cross_cats_sorted":[],"title_canon_sha256":"1736a3ccef88a48a1ae48506797b3b05598676b6a9cf4b6732e3ce79a52e339e","abstract_canon_sha256":"72f765768f27f975680997d2a5a9f8973ed7ffda2ce2502cef5e6112e7cb54d6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:28:50.159666Z","signature_b64":"KPBiEByIUcpTgGbqxLzV/VkYq5AlTsCjP2yDpAIdLEQtQzNaSIQTN04Zky3RSf4FA18SBH5XsAhKERwcCQpODg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"89b2ef7b034ff125ca487e3174b9ace5a7b58c723ba4dafefc7248be319822bd","last_reissued_at":"2026-07-05T09:28:50.159202Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:28:50.159202Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Grokked Transformers are Implicit Reasoners: A Mechanistic Journey to the Edge of Generalization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Boshi Wang, Huan Sun, Xiang Yue, Yu Su","submitted_at":"2024-05-23T21:42:19Z","abstract_excerpt":"We study whether transformers can learn to implicitly reason over parametric knowledge, a skill that even the most capable language models struggle with. Focusing on two representative reasoning types, composition and comparison, we consistently find that transformers can learn implicit reasoning, but only through grokking, i.e., extended training far beyond overfitting. The levels of generalization also vary across reasoning types: when faced with out-of-distribution examples, transformers fail to systematically generalize for composition but succeed for comparison. We delve into the model's "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.15071","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.15071/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.15071","created_at":"2026-07-05T09:28:50.159260+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.15071v3","created_at":"2026-07-05T09:28:50.159260+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.15071","created_at":"2026-07-05T09:28:50.159260+00:00"},{"alias_kind":"pith_short_12","alias_value":"RGZO66YDJ7YS","created_at":"2026-07-05T09:28:50.159260+00:00"},{"alias_kind":"pith_short_16","alias_value":"RGZO66YDJ7YSLSSI","created_at":"2026-07-05T09:28:50.159260+00:00"},{"alias_kind":"pith_short_8","alias_value":"RGZO66YD","created_at":"2026-07-05T09:28:50.159260+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20737","citing_title":"Repeated Shared Access Enables Grokking, but Edit Propagation Depends on an Addressable Memory","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00341","citing_title":"DiscoLoop: Looping Discrete Embeddings and Continuous Hidden States for Multi-hop Reasoning","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2510.01685","citing_title":"How Do Language Models Compose Functions?","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12394","citing_title":"Detecting overfitting in Neural Networks during long-horizon grokking using Random Matrix Theory","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13082","citing_title":"The Long Delay to Arithmetic Generalization: When Learned Representations Outrun Behavior","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13329","citing_title":"Tracing Persona Vectors Through LLM Pretraining","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12394","citing_title":"Detecting overfitting in Neural Networks during long-horizon grokking using Random Matrix Theory","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09724","citing_title":"Model Capacity Determines Grokking through Competing Memorisation and Generalisation Speeds","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22951","citing_title":"The Power of Power Law: Asymmetry Enables Compositional Reasoning","ref_index":51,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RGZO66YDJ7YSLSSIPYYXJONM4W","json":"https://pith.science/pith/RGZO66YDJ7YSLSSIPYYXJONM4W.json","graph_json":"https://pith.science/api/pith-number/RGZO66YDJ7YSLSSIPYYXJONM4W/graph.json","events_json":"https://pith.science/api/pith-number/RGZO66YDJ7YSLSSIPYYXJONM4W/events.json","paper":"https://pith.science/paper/RGZO66YD"},"agent_actions":{"view_html":"https://pith.science/pith/RGZO66YDJ7YSLSSIPYYXJONM4W","download_json":"https://pith.science/pith/RGZO66YDJ7YSLSSIPYYXJONM4W.json","view_paper":"https://pith.science/paper/RGZO66YD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.15071&json=true","fetch_graph":"https://pith.science/api/pith-number/RGZO66YDJ7YSLSSIPYYXJONM4W/graph.json","fetch_events":"https://pith.science/api/pith-number/RGZO66YDJ7YSLSSIPYYXJONM4W/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RGZO66YDJ7YSLSSIPYYXJONM4W/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RGZO66YDJ7YSLSSIPYYXJONM4W/action/storage_attestation","attest_author":"https://pith.science/pith/RGZO66YDJ7YSLSSIPYYXJONM4W/action/author_attestation","sign_citation":"https://pith.science/pith/RGZO66YDJ7YSLSSIPYYXJONM4W/action/citation_signature","submit_replication":"https://pith.science/pith/RGZO66YDJ7YSLSSIPYYXJONM4W/action/replication_record"}},"created_at":"2026-07-05T09:28:50.159260+00:00","updated_at":"2026-07-05T09:28:50.159260+00:00"}