{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:FNONOHGW5XO5EWHLNT2RXUIXKY","short_pith_number":"pith:FNONOHGW","schema_version":"1.0","canonical_sha256":"2b5cd71cd6edddd258eb6cf51bd11756232b0461529f2636b32a400ed4f0adf6","source":{"kind":"arxiv","id":"2406.03495","version":1},"attestation_state":"computed","paper":{"title":"Grokking Modular Polynomials","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cond-mat.dis-nn","hep-th","math.NT","stat.ML"],"primary_cat":"cs.LG","authors_text":"Andrey Gromov, Aritra Das, Darshil Doshi, Tianyu He","submitted_at":"2024-06-05T17:59:35Z","abstract_excerpt":"Neural networks readily learn a subset of the modular arithmetic tasks, while failing to generalize on the rest. This limitation remains unmoved by the choice of architecture and training strategies. On the other hand, an analytical solution for the weights of Multi-layer Perceptron (MLP) networks that generalize on the modular addition task is known in the literature. In this work, we (i) extend the class of analytical solutions to include modular multiplication as well as modular addition with many terms. Additionally, we show that real networks trained on these datasets learn similar soluti"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.03495","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-06-05T17:59:35Z","cross_cats_sorted":["cond-mat.dis-nn","hep-th","math.NT","stat.ML"],"title_canon_sha256":"b91dc35c2a3bf591450af93d491e91517d7d4b416802f47fd65b85dc886b5d74","abstract_canon_sha256":"a300deb3928726913a511854385b2644d4c6fae8e88bdd0a392be06a98b18f64"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:27:58.396557Z","signature_b64":"Bv7PuR0JHpNThNBdwrz+Ao5+qvrMYnoQRu/IBF7KdbAkQAw9UGCNXSgPzVE6sgqBWPDYevH/eJy38zo6fEbtBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2b5cd71cd6edddd258eb6cf51bd11756232b0461529f2636b32a400ed4f0adf6","last_reissued_at":"2026-07-05T08:27:58.396051Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:27:58.396051Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Grokking Modular Polynomials","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cond-mat.dis-nn","hep-th","math.NT","stat.ML"],"primary_cat":"cs.LG","authors_text":"Andrey Gromov, Aritra Das, Darshil Doshi, Tianyu He","submitted_at":"2024-06-05T17:59:35Z","abstract_excerpt":"Neural networks readily learn a subset of the modular arithmetic tasks, while failing to generalize on the rest. This limitation remains unmoved by the choice of architecture and training strategies. On the other hand, an analytical solution for the weights of Multi-layer Perceptron (MLP) networks that generalize on the modular addition task is known in the literature. In this work, we (i) extend the class of analytical solutions to include modular multiplication as well as modular addition with many terms. Additionally, we show that real networks trained on these datasets learn similar soluti"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.03495","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.03495/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.03495","created_at":"2026-07-05T08:27:58.396102+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.03495v1","created_at":"2026-07-05T08:27:58.396102+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.03495","created_at":"2026-07-05T08:27:58.396102+00:00"},{"alias_kind":"pith_short_12","alias_value":"FNONOHGW5XO5","created_at":"2026-07-05T08:27:58.396102+00:00"},{"alias_kind":"pith_short_16","alias_value":"FNONOHGW5XO5EWHL","created_at":"2026-07-05T08:27:58.396102+00:00"},{"alias_kind":"pith_short_8","alias_value":"FNONOHGW","created_at":"2026-07-05T08:27:58.396102+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.17399","citing_title":"The Discrete-Log Clock: How a Transformer Learns Modular Multiplication","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05863","citing_title":"Deciphering Two Training Clocks in Grokking via Deep Linear Network Theory with Conditional ReLU Reduction","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29548","citing_title":"Why Larger Models Learn More: Effects of Capacity, Interference, and Rare-Task Retention","ref_index":103,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17767","citing_title":"Feature Learning in Linear-Width Two-Layer Networks: Two vs. One Step of Gradient Descent","ref_index":199,"is_internal_anchor":false},{"citing_arxiv_id":"2510.04930","citing_title":"Egalitarian Gradient Descent: A Simple Approach to Accelerated Grokking","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17767","citing_title":"Feature Learning in Linear-Width Two-Layer Networks: Two vs. One Step of Gradient Descent","ref_index":199,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07648","citing_title":"Learning Large-Scale Modular Addition with an Auxiliary Modulus","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FNONOHGW5XO5EWHLNT2RXUIXKY","json":"https://pith.science/pith/FNONOHGW5XO5EWHLNT2RXUIXKY.json","graph_json":"https://pith.science/api/pith-number/FNONOHGW5XO5EWHLNT2RXUIXKY/graph.json","events_json":"https://pith.science/api/pith-number/FNONOHGW5XO5EWHLNT2RXUIXKY/events.json","paper":"https://pith.science/paper/FNONOHGW"},"agent_actions":{"view_html":"https://pith.science/pith/FNONOHGW5XO5EWHLNT2RXUIXKY","download_json":"https://pith.science/pith/FNONOHGW5XO5EWHLNT2RXUIXKY.json","view_paper":"https://pith.science/paper/FNONOHGW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.03495&json=true","fetch_graph":"https://pith.science/api/pith-number/FNONOHGW5XO5EWHLNT2RXUIXKY/graph.json","fetch_events":"https://pith.science/api/pith-number/FNONOHGW5XO5EWHLNT2RXUIXKY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FNONOHGW5XO5EWHLNT2RXUIXKY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FNONOHGW5XO5EWHLNT2RXUIXKY/action/storage_attestation","attest_author":"https://pith.science/pith/FNONOHGW5XO5EWHLNT2RXUIXKY/action/author_attestation","sign_citation":"https://pith.science/pith/FNONOHGW5XO5EWHLNT2RXUIXKY/action/citation_signature","submit_replication":"https://pith.science/pith/FNONOHGW5XO5EWHLNT2RXUIXKY/action/replication_record"}},"created_at":"2026-07-05T08:27:58.396102+00:00","updated_at":"2026-07-05T08:27:58.396102+00:00"}