{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:BFJHWNJT6THY5DKVQNRQ7HTDT5","short_pith_number":"pith:BFJHWNJT","schema_version":"1.0","canonical_sha256":"09527b3533f4cf8e8d5583630f9e639f7712ba4fe55b042e9338d967cbbec615","source":{"kind":"arxiv","id":"2403.02181","version":3},"attestation_state":"computed","paper":{"title":"Not All Layers of LLMs Are Necessary During Inference","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Aixin Sun, Peng Han, Shuo Shang, Siqi Fan, Xiang Li, Xin Jiang, Xuying Meng, Yequan Wang, Zhongyuan Wang","submitted_at":"2024-03-04T16:23:58Z","abstract_excerpt":"Due to the large number of parameters, the inference phase of Large Language Models (LLMs) is resource-intensive. However, not all requests posed to LLMs are equally difficult to handle. Through analysis, we show that for some tasks, LLMs can achieve results comparable to the final output at some intermediate layers. That is, not all layers of LLMs are necessary during inference. If we can predict at which layer the inferred results match the final results (produced by evaluating all layers), we could significantly reduce the inference cost. To this end, we propose a simple yet effective algor"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.02181","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-03-04T16:23:58Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"66925e3c7a17ea75c4998c888f94a6c5661855cd7097a0c7399e6f71d52f5e91","abstract_canon_sha256":"223dde983d3a15e326fe350ec697284595d9ce4b847622936f9775d9bce300ed"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:41:43.645248Z","signature_b64":"lKizOkMBre8NyG9EPzodT3ZoGFc6gSQpBO+zdV1FNjCKpKtquq/k0ST0nY6vmH8eiTAwCruP263EIEdMjpBpCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"09527b3533f4cf8e8d5583630f9e639f7712ba4fe55b042e9338d967cbbec615","last_reissued_at":"2026-07-05T08:41:43.644841Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:41:43.644841Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Not All Layers of LLMs Are Necessary During Inference","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Aixin Sun, Peng Han, Shuo Shang, Siqi Fan, Xiang Li, Xin Jiang, Xuying Meng, Yequan Wang, Zhongyuan Wang","submitted_at":"2024-03-04T16:23:58Z","abstract_excerpt":"Due to the large number of parameters, the inference phase of Large Language Models (LLMs) is resource-intensive. However, not all requests posed to LLMs are equally difficult to handle. Through analysis, we show that for some tasks, LLMs can achieve results comparable to the final output at some intermediate layers. That is, not all layers of LLMs are necessary during inference. If we can predict at which layer the inferred results match the final results (produced by evaluating all layers), we could significantly reduce the inference cost. To this end, we propose a simple yet effective algor"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.02181","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.02181/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.02181","created_at":"2026-07-05T08:41:43.644899+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.02181v3","created_at":"2026-07-05T08:41:43.644899+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.02181","created_at":"2026-07-05T08:41:43.644899+00:00"},{"alias_kind":"pith_short_12","alias_value":"BFJHWNJT6THY","created_at":"2026-07-05T08:41:43.644899+00:00"},{"alias_kind":"pith_short_16","alias_value":"BFJHWNJT6THY5DKV","created_at":"2026-07-05T08:41:43.644899+00:00"},{"alias_kind":"pith_short_8","alias_value":"BFJHWNJT","created_at":"2026-07-05T08:41:43.644899+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26620","citing_title":"Discovering Millions of Interpretable Features with Sparse Autoencoders","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06906","citing_title":"EASE-TTT: Evidence-Aligned Selective Test-Time Training for Long-Context Question Answering","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05586","citing_title":"BMCR: Adaptive Backbone Module Composition via Reinforcement Learning for Remote Sensing Object Detection","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27743","citing_title":"End-to-End Dynamic Sparsity for Resource-Adaptive LLM Inference","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27033","citing_title":"Tracing Computation Density in LLMs","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23033","citing_title":"Uncovering the Latent Potential of Deep Intermediate Representations","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2507.05387","citing_title":"The Generalization Ridge: Information Flow in Natural Language Generation","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18592","citing_title":"Two-dimensional early exit optimisation of LLM inference","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06393","citing_title":"ART: Attention Replacement Technique to Improve Factuality in LLMs","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20503","citing_title":"FASER: Fine-Grained Phase Management for Speculative Decoding in Dynamic LLM Serving","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BFJHWNJT6THY5DKVQNRQ7HTDT5","json":"https://pith.science/pith/BFJHWNJT6THY5DKVQNRQ7HTDT5.json","graph_json":"https://pith.science/api/pith-number/BFJHWNJT6THY5DKVQNRQ7HTDT5/graph.json","events_json":"https://pith.science/api/pith-number/BFJHWNJT6THY5DKVQNRQ7HTDT5/events.json","paper":"https://pith.science/paper/BFJHWNJT"},"agent_actions":{"view_html":"https://pith.science/pith/BFJHWNJT6THY5DKVQNRQ7HTDT5","download_json":"https://pith.science/pith/BFJHWNJT6THY5DKVQNRQ7HTDT5.json","view_paper":"https://pith.science/paper/BFJHWNJT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.02181&json=true","fetch_graph":"https://pith.science/api/pith-number/BFJHWNJT6THY5DKVQNRQ7HTDT5/graph.json","fetch_events":"https://pith.science/api/pith-number/BFJHWNJT6THY5DKVQNRQ7HTDT5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BFJHWNJT6THY5DKVQNRQ7HTDT5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BFJHWNJT6THY5DKVQNRQ7HTDT5/action/storage_attestation","attest_author":"https://pith.science/pith/BFJHWNJT6THY5DKVQNRQ7HTDT5/action/author_attestation","sign_citation":"https://pith.science/pith/BFJHWNJT6THY5DKVQNRQ7HTDT5/action/citation_signature","submit_replication":"https://pith.science/pith/BFJHWNJT6THY5DKVQNRQ7HTDT5/action/replication_record"}},"created_at":"2026-07-05T08:41:43.644899+00:00","updated_at":"2026-07-05T08:41:43.644899+00:00"}