{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JHBTHPERJSLW73X5P4MDIO7RHL","short_pith_number":"pith:JHBTHPER","schema_version":"1.0","canonical_sha256":"49c333bc914c976feefd7f18343bf13add20dd7b55f9e4096941cce680658ffc","source":{"kind":"arxiv","id":"2410.02984","version":1},"attestation_state":"computed","paper":{"title":"Differentiation and Specialization of Attention Heads via the Refined Local Learning Coefficient","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Daniel Murfet, George Wang, Jesse Hoogland, Stan van Wingerden, Zach Furman","submitted_at":"2024-10-03T20:51:02Z","abstract_excerpt":"We introduce refined variants of the Local Learning Coefficient (LLC), a measure of model complexity grounded in singular learning theory, to study the development of internal structure in transformer language models during training. By applying these \\textit{refined LLCs} (rLLCs) to individual components of a two-layer attention-only transformer, we gain novel insights into the progressive differentiation and specialization of attention heads. Our methodology reveals how attention heads differentiate into distinct functional roles over the course of training, analyzes the types of data these "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.02984","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-03T20:51:02Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"f18912a91f56cf32df4bd78d45cd3a9d6e5c531d02851eb6502710857afb23da","abstract_canon_sha256":"9144b584c4f297073794f93d0d78ac5783c07ead8ccf0604747e262b94950949"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:15:42.534368Z","signature_b64":"A/K1uKFMxlqzz51f+rVg8418KPz26tI3GkfASC9obdsRTznDgutuVXReCpZ7y+7SfsEzS8GI7N8JRxZo6AxeAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"49c333bc914c976feefd7f18343bf13add20dd7b55f9e4096941cce680658ffc","last_reissued_at":"2026-07-05T09:15:42.533949Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:15:42.533949Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Differentiation and Specialization of Attention Heads via the Refined Local Learning Coefficient","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Daniel Murfet, George Wang, Jesse Hoogland, Stan van Wingerden, Zach Furman","submitted_at":"2024-10-03T20:51:02Z","abstract_excerpt":"We introduce refined variants of the Local Learning Coefficient (LLC), a measure of model complexity grounded in singular learning theory, to study the development of internal structure in transformer language models during training. By applying these \\textit{refined LLCs} (rLLCs) to individual components of a two-layer attention-only transformer, we gain novel insights into the progressive differentiation and specialization of attention heads. Our methodology reveals how attention heads differentiate into distinct functional roles over the course of training, analyzes the types of data these "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.02984","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.02984/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.02984","created_at":"2026-07-05T09:15:42.534008+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.02984v1","created_at":"2026-07-05T09:15:42.534008+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.02984","created_at":"2026-07-05T09:15:42.534008+00:00"},{"alias_kind":"pith_short_12","alias_value":"JHBTHPERJSLW","created_at":"2026-07-05T09:15:42.534008+00:00"},{"alias_kind":"pith_short_16","alias_value":"JHBTHPERJSLW73X5","created_at":"2026-07-05T09:15:42.534008+00:00"},{"alias_kind":"pith_short_8","alias_value":"JHBTHPER","created_at":"2026-07-05T09:15:42.534008+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26050","citing_title":"Natural Ungrokking: Asymmetric Control of Which Rules Survive Pretraining","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21158","citing_title":"Dead-Direction Signatures: A Cheap Spectral Reading of Singular Complexity","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19491","citing_title":"Algebraic Dead Directions in LayerNorm Transformers: A Forward-Pass-Only Diagnostic at LLM Scale","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00603","citing_title":"Measuring Dead Directions: Decomposing and Classifying Singular Structure off Canonical Alignment","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05957","citing_title":"Dead Directions: Geometric Singular Learning","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20441","citing_title":"Weight Decay Regimes in Grokking Transformers: Cheap Online Diagnostics","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2509.24328","citing_title":"Speculative Verification: Exploiting Information Gain to Refine Speculative Decoding","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15183","citing_title":"When Are Two Networks the Same? Tensor Similarity for Mechanistic Interpretability","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18970","citing_title":"Mechanistic Anomaly Detection via Functional Attribution","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07980","citing_title":"Susceptibilities and Patterning: A Primer on Linear Response in Bayesian Learning","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JHBTHPERJSLW73X5P4MDIO7RHL","json":"https://pith.science/pith/JHBTHPERJSLW73X5P4MDIO7RHL.json","graph_json":"https://pith.science/api/pith-number/JHBTHPERJSLW73X5P4MDIO7RHL/graph.json","events_json":"https://pith.science/api/pith-number/JHBTHPERJSLW73X5P4MDIO7RHL/events.json","paper":"https://pith.science/paper/JHBTHPER"},"agent_actions":{"view_html":"https://pith.science/pith/JHBTHPERJSLW73X5P4MDIO7RHL","download_json":"https://pith.science/pith/JHBTHPERJSLW73X5P4MDIO7RHL.json","view_paper":"https://pith.science/paper/JHBTHPER","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.02984&json=true","fetch_graph":"https://pith.science/api/pith-number/JHBTHPERJSLW73X5P4MDIO7RHL/graph.json","fetch_events":"https://pith.science/api/pith-number/JHBTHPERJSLW73X5P4MDIO7RHL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JHBTHPERJSLW73X5P4MDIO7RHL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JHBTHPERJSLW73X5P4MDIO7RHL/action/storage_attestation","attest_author":"https://pith.science/pith/JHBTHPERJSLW73X5P4MDIO7RHL/action/author_attestation","sign_citation":"https://pith.science/pith/JHBTHPERJSLW73X5P4MDIO7RHL/action/citation_signature","submit_replication":"https://pith.science/pith/JHBTHPERJSLW73X5P4MDIO7RHL/action/replication_record"}},"created_at":"2026-07-05T09:15:42.534008+00:00","updated_at":"2026-07-05T09:15:42.534008+00:00"}