{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:SNOBE5BAV4HNPIMNKHKWQRFSLA","short_pith_number":"pith:SNOBE5BA","schema_version":"1.0","canonical_sha256":"935c127420af0ed7a18d51d56844b2580e21e4cab2548509fe357d82560e7013","source":{"kind":"arxiv","id":"2406.19384","version":3},"attestation_state":"computed","paper":{"title":"The Remarkable Robustness of LLMs: Stages of Inference?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Jin Hwa Lee, Max Tegmark, Vedang Lad, Wes Gurnee","submitted_at":"2024-06-27T17:57:03Z","abstract_excerpt":"We investigate the robustness of Large Language Models (LLMs) to structural interventions by deleting and swapping adjacent layers during inference. Surprisingly, models retain 72-95% of their original top-1 prediction accuracy without any fine-tuning. We find that performance degradation is not uniform across layers: interventions to the early and final layers cause the most degradation, while the model is remarkably robust to dropping middle layers. This pattern of localized sensitivity motivates our hypothesis of four stages of inference, observed across diverse model families and sizes: (1"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.19384","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-06-27T17:57:03Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"da245533fd7918493852d1a9f5473d7f6cbf6a2b91a0edb26051a12cb2598083","abstract_canon_sha256":"e9e17a269947ade0186cd0f141c102549f3656cb03174641c21504e6f99e46bf"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:21:33.739301Z","signature_b64":"r6o5I0oBnsDQ6GRVZpqdDSA19AtPyd0IN3NtiN3zgi/x+0+BTRmBiScwe2r6OQ3MUZb3sTz4vSk9PHIV+4rpDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"935c127420af0ed7a18d51d56844b2580e21e4cab2548509fe357d82560e7013","last_reissued_at":"2026-07-05T11:21:33.738720Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:21:33.738720Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Remarkable Robustness of LLMs: Stages of Inference?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Jin Hwa Lee, Max Tegmark, Vedang Lad, Wes Gurnee","submitted_at":"2024-06-27T17:57:03Z","abstract_excerpt":"We investigate the robustness of Large Language Models (LLMs) to structural interventions by deleting and swapping adjacent layers during inference. Surprisingly, models retain 72-95% of their original top-1 prediction accuracy without any fine-tuning. We find that performance degradation is not uniform across layers: interventions to the early and final layers cause the most degradation, while the model is remarkably robust to dropping middle layers. This pattern of localized sensitivity motivates our hypothesis of four stages of inference, observed across diverse model families and sizes: (1"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.19384","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.19384/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.19384","created_at":"2026-07-05T11:21:33.738792+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.19384v3","created_at":"2026-07-05T11:21:33.738792+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.19384","created_at":"2026-07-05T11:21:33.738792+00:00"},{"alias_kind":"pith_short_12","alias_value":"SNOBE5BAV4HN","created_at":"2026-07-05T11:21:33.738792+00:00"},{"alias_kind":"pith_short_16","alias_value":"SNOBE5BAV4HNPIMN","created_at":"2026-07-05T11:21:33.738792+00:00"},{"alias_kind":"pith_short_8","alias_value":"SNOBE5BA","created_at":"2026-07-05T11:21:33.738792+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23670","citing_title":"Tapered Language Models","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08562","citing_title":"Inside the LLM Word Factory","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30813","citing_title":"Gradient Smoothing: Coupling Layer-wise Updates for Improved Optimization","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07604","citing_title":"Contribution Weights: A Geometrical Analysis of Self-Attention Transformers","ref_index":82,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23033","citing_title":"Uncovering the Latent Potential of Deep Intermediate Representations","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23872","citing_title":"Training-Free Looped Transformers","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2602.01997","citing_title":"On the Limits of Layer Pruning for Generative Reasoning in Large Language Models","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2602.21750","citing_title":"From Words to Amino Acids: Does the Curse of Depth Persist?","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19052","citing_title":"Cell-Based Representation of Relational Binding in Language Models","ref_index":72,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12426","citing_title":"Do Transformers Use their Depth Adaptively? Evidence from a Relational Reasoning Task","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11791","citing_title":"A Mechanistic Analysis of Looped Reasoning Language Models","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05223","citing_title":"Structural Instability of Feature Composition","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SNOBE5BAV4HNPIMNKHKWQRFSLA","json":"https://pith.science/pith/SNOBE5BAV4HNPIMNKHKWQRFSLA.json","graph_json":"https://pith.science/api/pith-number/SNOBE5BAV4HNPIMNKHKWQRFSLA/graph.json","events_json":"https://pith.science/api/pith-number/SNOBE5BAV4HNPIMNKHKWQRFSLA/events.json","paper":"https://pith.science/paper/SNOBE5BA"},"agent_actions":{"view_html":"https://pith.science/pith/SNOBE5BAV4HNPIMNKHKWQRFSLA","download_json":"https://pith.science/pith/SNOBE5BAV4HNPIMNKHKWQRFSLA.json","view_paper":"https://pith.science/paper/SNOBE5BA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.19384&json=true","fetch_graph":"https://pith.science/api/pith-number/SNOBE5BAV4HNPIMNKHKWQRFSLA/graph.json","fetch_events":"https://pith.science/api/pith-number/SNOBE5BAV4HNPIMNKHKWQRFSLA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SNOBE5BAV4HNPIMNKHKWQRFSLA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SNOBE5BAV4HNPIMNKHKWQRFSLA/action/storage_attestation","attest_author":"https://pith.science/pith/SNOBE5BAV4HNPIMNKHKWQRFSLA/action/author_attestation","sign_citation":"https://pith.science/pith/SNOBE5BAV4HNPIMNKHKWQRFSLA/action/citation_signature","submit_replication":"https://pith.science/pith/SNOBE5BAV4HNPIMNKHKWQRFSLA/action/replication_record"}},"created_at":"2026-07-05T11:21:33.738792+00:00","updated_at":"2026-07-05T11:21:33.738792+00:00"}