{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:LT65TQ4CRRLNR36LPIX6UGT6FM","short_pith_number":"pith:LT65TQ4C","schema_version":"1.0","canonical_sha256":"5cfdd9c3828c56d8efcb7a2fea1a7e2b34126fc7eae375eb14d7d68c4c1bf180","source":{"kind":"arxiv","id":"2302.10322","version":1},"attestation_state":"computed","paper":{"title":"Deep Transformers without Shortcuts: Modifying Self-attention for Faithful Signal Propagation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","stat.ML"],"primary_cat":"cs.LG","authors_text":"Aleksandar Botev, Andrew Brock, Bobby He, Guodong Zhang, James Martens, Samuel L Smith, Yee Whye Teh","submitted_at":"2023-02-20T21:26:25Z","abstract_excerpt":"Skip connections and normalisation layers form two standard architectural components that are ubiquitous for the training of Deep Neural Networks (DNNs), but whose precise roles are poorly understood. Recent approaches such as Deep Kernel Shaping have made progress towards reducing our reliance on them, using insights from wide NN kernel theory to improve signal propagation in vanilla DNNs (which we define as networks without skips or normalisation). However, these approaches are incompatible with the self-attention layers present in transformers, whose kernels are intrinsically more complicat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2302.10322","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-02-20T21:26:25Z","cross_cats_sorted":["cs.AI","cs.CL","stat.ML"],"title_canon_sha256":"d7b97513539c50481c3873aa9b7d3c1cd6df426fb5a875d743ab489ee480a091","abstract_canon_sha256":"41be5e7b63050e8e58ab5b7e96aecb3c568f98211a26622d42fc09889e8ebac5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:44:16.908245Z","signature_b64":"KpizqQvSbVamsNydtDw0tlxE9HA4QijOJaMbBG4vCs+B5OFxgH+EfpR1f3v8W60TbSRv+0MNd4Ymwg6dmnzjBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5cfdd9c3828c56d8efcb7a2fea1a7e2b34126fc7eae375eb14d7d68c4c1bf180","last_reissued_at":"2026-07-05T05:44:16.907796Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:44:16.907796Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Deep Transformers without Shortcuts: Modifying Self-attention for Faithful Signal Propagation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","stat.ML"],"primary_cat":"cs.LG","authors_text":"Aleksandar Botev, Andrew Brock, Bobby He, Guodong Zhang, James Martens, Samuel L Smith, Yee Whye Teh","submitted_at":"2023-02-20T21:26:25Z","abstract_excerpt":"Skip connections and normalisation layers form two standard architectural components that are ubiquitous for the training of Deep Neural Networks (DNNs), but whose precise roles are poorly understood. Recent approaches such as Deep Kernel Shaping have made progress towards reducing our reliance on them, using insights from wide NN kernel theory to improve signal propagation in vanilla DNNs (which we define as networks without skips or normalisation). However, these approaches are incompatible with the self-attention layers present in transformers, whose kernels are intrinsically more complicat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2302.10322","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2302.10322/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2302.10322","created_at":"2026-07-05T05:44:16.907861+00:00"},{"alias_kind":"arxiv_version","alias_value":"2302.10322v1","created_at":"2026-07-05T05:44:16.907861+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2302.10322","created_at":"2026-07-05T05:44:16.907861+00:00"},{"alias_kind":"pith_short_12","alias_value":"LT65TQ4CRRLN","created_at":"2026-07-05T05:44:16.907861+00:00"},{"alias_kind":"pith_short_16","alias_value":"LT65TQ4CRRLNR36L","created_at":"2026-07-05T05:44:16.907861+00:00"},{"alias_kind":"pith_short_8","alias_value":"LT65TQ4C","created_at":"2026-07-05T05:44:16.907861+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2608.09417","citing_title":"Why Post-Norm Transformers Collapse: Attention Amplification and Gradient Repair Failure","ref_index":44,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LT65TQ4CRRLNR36LPIX6UGT6FM","json":"https://pith.science/pith/LT65TQ4CRRLNR36LPIX6UGT6FM.json","graph_json":"https://pith.science/api/pith-number/LT65TQ4CRRLNR36LPIX6UGT6FM/graph.json","events_json":"https://pith.science/api/pith-number/LT65TQ4CRRLNR36LPIX6UGT6FM/events.json","paper":"https://pith.science/paper/LT65TQ4C"},"agent_actions":{"view_html":"https://pith.science/pith/LT65TQ4CRRLNR36LPIX6UGT6FM","download_json":"https://pith.science/pith/LT65TQ4CRRLNR36LPIX6UGT6FM.json","view_paper":"https://pith.science/paper/LT65TQ4C","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2302.10322&json=true","fetch_graph":"https://pith.science/api/pith-number/LT65TQ4CRRLNR36LPIX6UGT6FM/graph.json","fetch_events":"https://pith.science/api/pith-number/LT65TQ4CRRLNR36LPIX6UGT6FM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LT65TQ4CRRLNR36LPIX6UGT6FM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LT65TQ4CRRLNR36LPIX6UGT6FM/action/storage_attestation","attest_author":"https://pith.science/pith/LT65TQ4CRRLNR36LPIX6UGT6FM/action/author_attestation","sign_citation":"https://pith.science/pith/LT65TQ4CRRLNR36LPIX6UGT6FM/action/citation_signature","submit_replication":"https://pith.science/pith/LT65TQ4CRRLNR36LPIX6UGT6FM/action/replication_record"}},"created_at":"2026-07-05T05:44:16.907861+00:00","updated_at":"2026-07-05T05:44:16.907861+00:00"}