{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:43NMRLZKWKLTAW44GHYX27HY4T","short_pith_number":"pith:43NMRLZK","schema_version":"1.0","canonical_sha256":"e6dac8af2ab297305b9c31f17d7cf8e4d69395a705037cc1e6b19af4f8f8b596","source":{"kind":"arxiv","id":"2410.24050","version":3},"attestation_state":"computed","paper":{"title":"A Mechanistic Study of Transformers Training Dynamics","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Ambroise Odonnat, Vivien Cabannes, Wassim Bouaziz","submitted_at":"2024-10-31T15:46:10Z","abstract_excerpt":"Large-scale pretraining of transformers has been central to the success of foundation models. However, the scale of those models limits our understanding of the mechanisms at play during optimization. In this work, we study the training dynamics of transformers in a controlled and interpretable setting. On the sparse modular addition task, we demonstrate that specialized attention circuits, called clustering heads, can be implemented during gradient descent to solve the problem. Our experiments show that such pathways naturally emerge during training. By monitoring the evolution of tokens via "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.24050","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-31T15:46:10Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"a2135071acbcab095959a83b7537b4ebf9534369c36f34bced02b634483704e9","abstract_canon_sha256":"34d575ea2e50906b5ac7d307ce6b4d580d9522635d9df3916cd2325662ccae5a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-06-30T02:18:03.558818Z","signature_b64":"5lTXDTYLsnbRKfM8WR15heywXBm0osG9hEJn3YGMsutZygqsFG/jiWd1/PE32+utwIL9bMvyEX7WJ9nG/8YcAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e6dac8af2ab297305b9c31f17d7cf8e4d69395a705037cc1e6b19af4f8f8b596","last_reissued_at":"2026-06-30T02:18:03.558145Z","signature_status":"signed_v1","first_computed_at":"2026-06-30T02:18:03.558145Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Mechanistic Study of Transformers Training Dynamics","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Ambroise Odonnat, Vivien Cabannes, Wassim Bouaziz","submitted_at":"2024-10-31T15:46:10Z","abstract_excerpt":"Large-scale pretraining of transformers has been central to the success of foundation models. However, the scale of those models limits our understanding of the mechanisms at play during optimization. In this work, we study the training dynamics of transformers in a controlled and interpretable setting. On the sparse modular addition task, we demonstrate that specialized attention circuits, called clustering heads, can be implemented during gradient descent to solve the problem. Our experiments show that such pathways naturally emerge during training. By monitoring the evolution of tokens via "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.24050","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.24050/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.24050","created_at":"2026-06-30T02:18:03.558235+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.24050v3","created_at":"2026-06-30T02:18:03.558235+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.24050","created_at":"2026-06-30T02:18:03.558235+00:00"},{"alias_kind":"pith_short_12","alias_value":"43NMRLZKWKLT","created_at":"2026-06-30T02:18:03.558235+00:00"},{"alias_kind":"pith_short_16","alias_value":"43NMRLZKWKLTAW44","created_at":"2026-06-30T02:18:03.558235+00:00"},{"alias_kind":"pith_short_8","alias_value":"43NMRLZK","created_at":"2026-06-30T02:18:03.558235+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.12837","citing_title":"Learning In-context n-grams with Transformers: Sub-n-grams Are Near-stationary Points","ref_index":2025,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/43NMRLZKWKLTAW44GHYX27HY4T","json":"https://pith.science/pith/43NMRLZKWKLTAW44GHYX27HY4T.json","graph_json":"https://pith.science/api/pith-number/43NMRLZKWKLTAW44GHYX27HY4T/graph.json","events_json":"https://pith.science/api/pith-number/43NMRLZKWKLTAW44GHYX27HY4T/events.json","paper":"https://pith.science/paper/43NMRLZK"},"agent_actions":{"view_html":"https://pith.science/pith/43NMRLZKWKLTAW44GHYX27HY4T","download_json":"https://pith.science/pith/43NMRLZKWKLTAW44GHYX27HY4T.json","view_paper":"https://pith.science/paper/43NMRLZK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.24050&json=true","fetch_graph":"https://pith.science/api/pith-number/43NMRLZKWKLTAW44GHYX27HY4T/graph.json","fetch_events":"https://pith.science/api/pith-number/43NMRLZKWKLTAW44GHYX27HY4T/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/43NMRLZKWKLTAW44GHYX27HY4T/action/timestamp_anchor","attest_storage":"https://pith.science/pith/43NMRLZKWKLTAW44GHYX27HY4T/action/storage_attestation","attest_author":"https://pith.science/pith/43NMRLZKWKLTAW44GHYX27HY4T/action/author_attestation","sign_citation":"https://pith.science/pith/43NMRLZKWKLTAW44GHYX27HY4T/action/citation_signature","submit_replication":"https://pith.science/pith/43NMRLZKWKLTAW44GHYX27HY4T/action/replication_record"}},"created_at":"2026-06-30T02:18:03.558235+00:00","updated_at":"2026-06-30T02:18:03.558235+00:00"}