{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:45X6G64LUGH3QRLGAB6CMWGW4I","short_pith_number":"pith:45X6G64L","schema_version":"1.0","canonical_sha256":"e76fe37b8ba18fb84566007c2658d6e21e5949558eb3bbfab9a930afc31dfd30","source":{"kind":"arxiv","id":"2506.21103","version":1},"attestation_state":"computed","paper":{"title":"Learning to Skip the Middle Layers of Transformers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Laurence Aitchison, Tim Lawson","submitted_at":"2025-06-26T09:01:19Z","abstract_excerpt":"Conditional computation is a popular strategy to make Transformers more efficient. Existing methods often target individual modules (e.g., mixture-of-experts layers) or skip layers independently of one another. However, interpretability research has demonstrated that the middle layers of Transformers exhibit greater redundancy, and that early layers aggregate information into token positions. Guided by these insights, we propose a novel architecture that dynamically skips a variable number of layers from the middle outward. In particular, a learned gating mechanism determines whether to bypass"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.21103","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-26T09:01:19Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"c4984b874341d1b7dfd38f31e9eb9855c03f06e57f06469ecf626be83d1b429c","abstract_canon_sha256":"4509d8e2eddcdaaeac35d71884c423477bad82affd92ba5a39d344259b323928"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:27:40.164761Z","signature_b64":"2TQn0VJ9nrqzzRLE9WcT2ME5qQJsv3SwKd/lKkyn71T6H5EH8hbsyA1d0nGENwjm6ry9ckNDVrA3rY7HVtGDCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e76fe37b8ba18fb84566007c2658d6e21e5949558eb3bbfab9a930afc31dfd30","last_reissued_at":"2026-07-05T11:27:40.164323Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:27:40.164323Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning to Skip the Middle Layers of Transformers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Laurence Aitchison, Tim Lawson","submitted_at":"2025-06-26T09:01:19Z","abstract_excerpt":"Conditional computation is a popular strategy to make Transformers more efficient. Existing methods often target individual modules (e.g., mixture-of-experts layers) or skip layers independently of one another. However, interpretability research has demonstrated that the middle layers of Transformers exhibit greater redundancy, and that early layers aggregate information into token positions. Guided by these insights, we propose a novel architecture that dynamically skips a variable number of layers from the middle outward. In particular, a learned gating mechanism determines whether to bypass"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.21103","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.21103/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.21103","created_at":"2026-07-05T11:27:40.164382+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.21103v1","created_at":"2026-07-05T11:27:40.164382+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.21103","created_at":"2026-07-05T11:27:40.164382+00:00"},{"alias_kind":"pith_short_12","alias_value":"45X6G64LUGH3","created_at":"2026-07-05T11:27:40.164382+00:00"},{"alias_kind":"pith_short_16","alias_value":"45X6G64LUGH3QRLG","created_at":"2026-07-05T11:27:40.164382+00:00"},{"alias_kind":"pith_short_8","alias_value":"45X6G64L","created_at":"2026-07-05T11:27:40.164382+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06631","citing_title":"Dynamic-in-Few-Step: Unifying Dynamic Computation and Few-Step Distillation for Efficient Video Generation","ref_index":18,"is_internal_anchor":true},{"citing_arxiv_id":"2606.12412","citing_title":"Reroute, Don't Remove: Recoverable Visual Token Routing for Vision-Language Models","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14738","citing_title":"TAPIOCA: Why Task- Aware Pruning Improves OOD model Capability","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2601.14004","citing_title":"Locate, Steer, and Improve: A Practical Survey of Actionable Mechanistic Interpretability in Large Language Models","ref_index":161,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14438","citing_title":"BEAM: Binary Expert Activation Masking for Dynamic Routing in MoE","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22583","citing_title":"Adaptive Head Budgeting for Efficient Multi-Head Attention","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/45X6G64LUGH3QRLGAB6CMWGW4I","json":"https://pith.science/pith/45X6G64LUGH3QRLGAB6CMWGW4I.json","graph_json":"https://pith.science/api/pith-number/45X6G64LUGH3QRLGAB6CMWGW4I/graph.json","events_json":"https://pith.science/api/pith-number/45X6G64LUGH3QRLGAB6CMWGW4I/events.json","paper":"https://pith.science/paper/45X6G64L"},"agent_actions":{"view_html":"https://pith.science/pith/45X6G64LUGH3QRLGAB6CMWGW4I","download_json":"https://pith.science/pith/45X6G64LUGH3QRLGAB6CMWGW4I.json","view_paper":"https://pith.science/paper/45X6G64L","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.21103&json=true","fetch_graph":"https://pith.science/api/pith-number/45X6G64LUGH3QRLGAB6CMWGW4I/graph.json","fetch_events":"https://pith.science/api/pith-number/45X6G64LUGH3QRLGAB6CMWGW4I/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/45X6G64LUGH3QRLGAB6CMWGW4I/action/timestamp_anchor","attest_storage":"https://pith.science/pith/45X6G64LUGH3QRLGAB6CMWGW4I/action/storage_attestation","attest_author":"https://pith.science/pith/45X6G64LUGH3QRLGAB6CMWGW4I/action/author_attestation","sign_citation":"https://pith.science/pith/45X6G64LUGH3QRLGAB6CMWGW4I/action/citation_signature","submit_replication":"https://pith.science/pith/45X6G64LUGH3QRLGAB6CMWGW4I/action/replication_record"}},"created_at":"2026-07-05T11:27:40.164382+00:00","updated_at":"2026-07-05T11:27:40.164382+00:00"}