{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:5F3ENJIGWIYABCADBJAOKW45BB","short_pith_number":"pith:5F3ENJIG","schema_version":"1.0","canonical_sha256":"e97646a506b2300088030a40e55b9d0844cfae822fb755ec33b54c34ae054d6c","source":{"kind":"arxiv","id":"2407.09298","version":4},"attestation_state":"computed","paper":{"title":"Transformer Layers as Painters","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Aakash Kumar Nain, Llion Jones, Marc Pickett, Qi Sun","submitted_at":"2024-07-12T14:31:05Z","abstract_excerpt":"Despite their nearly universal adoption for large language models, the internal workings of transformers are not well understood. We aim to better understand the impact of removing or reorganizing information throughout the layers of a pretrained transformer. Such an understanding could both yield better usage of existing models as well as to make architectural improvements to produce new variants. We present a series of empirical studies on frozen models that show that the lower and final layers of pretrained transformers differ from middle layers, but that middle layers have a surprising amo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.09298","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-07-12T14:31:05Z","cross_cats_sorted":[],"title_canon_sha256":"171db22f53fbf8ab8407bec87183f2626c5d1cf8703ca71471652b19a49074b9","abstract_canon_sha256":"60425359c3bcb9e19371e298d37059e2644ba67ee1d9eb7d5367d8f78bd63c94"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:13:26.670086Z","signature_b64":"SDVxzolaTlhdymfBHOZUTGUr9spbdSW29nDlGmChaDV8LRtTyHSKawSDNp7w4/AwShAby5qeMC9K2nvu9QVkBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e97646a506b2300088030a40e55b9d0844cfae822fb755ec33b54c34ae054d6c","last_reissued_at":"2026-07-05T10:13:26.669597Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:13:26.669597Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Transformer Layers as Painters","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Aakash Kumar Nain, Llion Jones, Marc Pickett, Qi Sun","submitted_at":"2024-07-12T14:31:05Z","abstract_excerpt":"Despite their nearly universal adoption for large language models, the internal workings of transformers are not well understood. We aim to better understand the impact of removing or reorganizing information throughout the layers of a pretrained transformer. Such an understanding could both yield better usage of existing models as well as to make architectural improvements to produce new variants. We present a series of empirical studies on frozen models that show that the lower and final layers of pretrained transformers differ from middle layers, but that middle layers have a surprising amo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.09298","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.09298/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.09298","created_at":"2026-07-05T10:13:26.669657+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.09298v4","created_at":"2026-07-05T10:13:26.669657+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.09298","created_at":"2026-07-05T10:13:26.669657+00:00"},{"alias_kind":"pith_short_12","alias_value":"5F3ENJIGWIYA","created_at":"2026-07-05T10:13:26.669657+00:00"},{"alias_kind":"pith_short_16","alias_value":"5F3ENJIGWIYABCAD","created_at":"2026-07-05T10:13:26.669657+00:00"},{"alias_kind":"pith_short_8","alias_value":"5F3ENJIG","created_at":"2026-07-05T10:13:26.669657+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09357","citing_title":"Rethinking Depth: A study of the Recursive-Transformer for Speech Recognition","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27705","citing_title":"Mitigating Position Bias in Transformers via Layer-Specific Positional Embedding Scaling","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2502.05171","citing_title":"Scaling up Test-Time Compute with Latent Reasoning: A Recurrent Depth Approach","ref_index":149,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06393","citing_title":"ART: Attention Replacement Technique to Improve Factuality in LLMs","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5F3ENJIGWIYABCADBJAOKW45BB","json":"https://pith.science/pith/5F3ENJIGWIYABCADBJAOKW45BB.json","graph_json":"https://pith.science/api/pith-number/5F3ENJIGWIYABCADBJAOKW45BB/graph.json","events_json":"https://pith.science/api/pith-number/5F3ENJIGWIYABCADBJAOKW45BB/events.json","paper":"https://pith.science/paper/5F3ENJIG"},"agent_actions":{"view_html":"https://pith.science/pith/5F3ENJIGWIYABCADBJAOKW45BB","download_json":"https://pith.science/pith/5F3ENJIGWIYABCADBJAOKW45BB.json","view_paper":"https://pith.science/paper/5F3ENJIG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.09298&json=true","fetch_graph":"https://pith.science/api/pith-number/5F3ENJIGWIYABCADBJAOKW45BB/graph.json","fetch_events":"https://pith.science/api/pith-number/5F3ENJIGWIYABCADBJAOKW45BB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5F3ENJIGWIYABCADBJAOKW45BB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5F3ENJIGWIYABCADBJAOKW45BB/action/storage_attestation","attest_author":"https://pith.science/pith/5F3ENJIGWIYABCADBJAOKW45BB/action/author_attestation","sign_citation":"https://pith.science/pith/5F3ENJIGWIYABCADBJAOKW45BB/action/citation_signature","submit_replication":"https://pith.science/pith/5F3ENJIGWIYABCADBJAOKW45BB/action/replication_record"}},"created_at":"2026-07-05T10:13:26.669657+00:00","updated_at":"2026-07-05T10:13:26.669657+00:00"}