{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:7ISGGDM6EYMXEBYAQQIGB4XEK4","short_pith_number":"pith:7ISGGDM6","schema_version":"1.0","canonical_sha256":"fa24630d9e2619720700841060f2e4571b197ce376421755a166146eb863bf27","source":{"kind":"arxiv","id":"2512.17351","version":2},"attestation_state":"computed","paper":{"title":"Physics of Language Models: Part 4.1, Architecture Design and the Magic of Canon Layers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Zeyuan Allen-Zhu","submitted_at":"2025-12-19T08:47:28Z","abstract_excerpt":"Understanding architectural differences in language models is challenging, especially at academic-scale pretraining (e.g., 1.3B parameters, 100B tokens), where results are often dominated by noise and randomness. To overcome this, we introduce controlled synthetic pretraining tasks that isolate and evaluate core model capabilities. Within this framework, we discover CANON LAYERS: lightweight architectural components -- named after the musical term \"canon\" -- that promote horizontal information flow across neighboring tokens. Canon layers compute weighted sums of nearby token representations an"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2512.17351","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-12-19T08:47:28Z","cross_cats_sorted":[],"title_canon_sha256":"9461a6669c50d9f6fd3f84238c636535ba940c895665d87e9413376d0cdb8f76","abstract_canon_sha256":"d557b8b55a8c5783e4bf269e1877ef4c3c50a14d4e91915434060670de8eaa2a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-29T01:25:32.350454Z","signature_b64":"lrhE5Ut04JHudeU1iIRLpwnp9/ZTjD60PSpw0vISMiWSCDxuTQkOzCeFpTuPpvH+peVXM6L74n3IY/cqgPR0CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fa24630d9e2619720700841060f2e4571b197ce376421755a166146eb863bf27","last_reissued_at":"2026-07-29T01:25:32.349463Z","signature_status":"signed_v1","first_computed_at":"2026-07-29T01:25:32.349463Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Physics of Language Models: Part 4.1, Architecture Design and the Magic of Canon Layers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Zeyuan Allen-Zhu","submitted_at":"2025-12-19T08:47:28Z","abstract_excerpt":"Understanding architectural differences in language models is challenging, especially at academic-scale pretraining (e.g., 1.3B parameters, 100B tokens), where results are often dominated by noise and randomness. To overcome this, we introduce controlled synthetic pretraining tasks that isolate and evaluate core model capabilities. Within this framework, we discover CANON LAYERS: lightweight architectural components -- named after the musical term \"canon\" -- that promote horizontal information flow across neighboring tokens. Canon layers compute weighted sums of nearby token representations an"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2512.17351","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2512.17351/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2512.17351","created_at":"2026-07-29T01:25:32.349937+00:00"},{"alias_kind":"arxiv_version","alias_value":"2512.17351v2","created_at":"2026-07-29T01:25:32.349937+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2512.17351","created_at":"2026-07-29T01:25:32.349937+00:00"},{"alias_kind":"pith_short_12","alias_value":"7ISGGDM6EYMX","created_at":"2026-07-29T01:25:32.349937+00:00"},{"alias_kind":"pith_short_16","alias_value":"7ISGGDM6EYMXEBYA","created_at":"2026-07-29T01:25:32.349937+00:00"},{"alias_kind":"pith_short_8","alias_value":"7ISGGDM6","created_at":"2026-07-29T01:25:32.349937+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":10,"sample":[{"citing_arxiv_id":"2607.07706","citing_title":"The Key to Going Linear: Analysis-Driven Transformer Linearization","ref_index":2,"is_internal_anchor":true},{"citing_arxiv_id":"2606.25342","citing_title":"Lifelong In-Context Learning with Transformers Requires Parametric Forms of Attention","ref_index":2,"is_internal_anchor":true},{"citing_arxiv_id":"2606.25156","citing_title":"ATMA: Length-Invariant Language Modeling via Polar Attention and Gated-Delta Compression Memory","ref_index":2,"is_internal_anchor":true},{"citing_arxiv_id":"2606.03825","citing_title":"Dynamic Short Convolutions Improve Transformers","ref_index":30,"is_internal_anchor":true},{"citing_arxiv_id":"2605.11287","citing_title":"Beyond Similarity: Temporal Operator Attention for Time Series Analysis","ref_index":1,"is_internal_anchor":true},{"citing_arxiv_id":"2606.25156","citing_title":"ATMA: Length-Invariant Language Modeling via Polar Attention and Gated-Delta Compression Memory","ref_index":2,"is_internal_anchor":true},{"citing_arxiv_id":"2606.29858","citing_title":"Smooth Scaling Laws Hide Stepwise Token Learning","ref_index":41,"is_internal_anchor":true},{"citing_arxiv_id":"2605.11287","citing_title":"Beyond Similarity: Temporal Operator Attention for Time Series Analysis","ref_index":1,"is_internal_anchor":true},{"citing_arxiv_id":"2604.17121","citing_title":"The Topological Trouble With Transformers","ref_index":1,"is_internal_anchor":true},{"citing_arxiv_id":"2604.20817","citing_title":"Convergent Evolution: How Different Language Models Learn Similar Number Representations","ref_index":35,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7ISGGDM6EYMXEBYAQQIGB4XEK4","json":"https://pith.science/pith/7ISGGDM6EYMXEBYAQQIGB4XEK4.json","graph_json":"https://pith.science/api/pith-number/7ISGGDM6EYMXEBYAQQIGB4XEK4/graph.json","events_json":"https://pith.science/api/pith-number/7ISGGDM6EYMXEBYAQQIGB4XEK4/events.json","paper":"https://pith.science/paper/7ISGGDM6"},"agent_actions":{"view_html":"https://pith.science/pith/7ISGGDM6EYMXEBYAQQIGB4XEK4","download_json":"https://pith.science/pith/7ISGGDM6EYMXEBYAQQIGB4XEK4.json","view_paper":"https://pith.science/paper/7ISGGDM6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2512.17351&json=true","fetch_graph":"https://pith.science/api/pith-number/7ISGGDM6EYMXEBYAQQIGB4XEK4/graph.json","fetch_events":"https://pith.science/api/pith-number/7ISGGDM6EYMXEBYAQQIGB4XEK4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7ISGGDM6EYMXEBYAQQIGB4XEK4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7ISGGDM6EYMXEBYAQQIGB4XEK4/action/storage_attestation","attest_author":"https://pith.science/pith/7ISGGDM6EYMXEBYAQQIGB4XEK4/action/author_attestation","sign_citation":"https://pith.science/pith/7ISGGDM6EYMXEBYAQQIGB4XEK4/action/citation_signature","submit_replication":"https://pith.science/pith/7ISGGDM6EYMXEBYAQQIGB4XEK4/action/replication_record"}},"created_at":"2026-07-29T01:25:32.349937+00:00","updated_at":"2026-07-29T01:25:32.349937+00:00"}