{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:VITLZQIEI6MJTI5PTBL2DQA26E","short_pith_number":"pith:VITLZQIE","schema_version":"1.0","canonical_sha256":"aa26bcc104479899a3af9857a1c01af10ec5f5d354da4634de1f82fb9990ccae","source":{"kind":"arxiv","id":"2409.14381","version":1},"attestation_state":"computed","paper":{"title":"Investigating Layer Importance in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Kenji Kawaguchi, Yanfei Dong, Yang Zhang","submitted_at":"2024-09-22T09:53:13Z","abstract_excerpt":"Large language models (LLMs) have gained increasing attention due to their prominent ability to understand and process texts. Nevertheless, LLMs largely remain opaque. The lack of understanding of LLMs has obstructed the deployment in safety-critical scenarios and hindered the development of better models. In this study, we advance the understanding of LLM by investigating the significance of individual layers in LLMs. We propose an efficient sampling method to faithfully evaluate the importance of layers using Shapley values, a widely used explanation framework in feature attribution and data"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.14381","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-09-22T09:53:13Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"cbb2bed4e44f24b9f4754f9e705055c359c1918e5848f5db88090c685c15a864","abstract_canon_sha256":"1f6363af75ae8ea0f4d3ed808a79280936b1ee82c904f5851ffb5b45bb82b708"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:10:15.844978Z","signature_b64":"D7tFzHedQyq8fROlfWGipQoCxjNld6r/0vmqTPJPGnUhS37KttV0x+niuB4gfwm7oD+FQFPirO/I/KAjlnBzCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"aa26bcc104479899a3af9857a1c01af10ec5f5d354da4634de1f82fb9990ccae","last_reissued_at":"2026-07-05T09:10:15.844519Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:10:15.844519Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Investigating Layer Importance in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Kenji Kawaguchi, Yanfei Dong, Yang Zhang","submitted_at":"2024-09-22T09:53:13Z","abstract_excerpt":"Large language models (LLMs) have gained increasing attention due to their prominent ability to understand and process texts. Nevertheless, LLMs largely remain opaque. The lack of understanding of LLMs has obstructed the deployment in safety-critical scenarios and hindered the development of better models. In this study, we advance the understanding of LLM by investigating the significance of individual layers in LLMs. We propose an efficient sampling method to faithfully evaluate the importance of layers using Shapley values, a widely used explanation framework in feature attribution and data"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.14381","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.14381/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.14381","created_at":"2026-07-05T09:10:15.844577+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.14381v1","created_at":"2026-07-05T09:10:15.844577+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.14381","created_at":"2026-07-05T09:10:15.844577+00:00"},{"alias_kind":"pith_short_12","alias_value":"VITLZQIEI6MJ","created_at":"2026-07-05T09:10:15.844577+00:00"},{"alias_kind":"pith_short_16","alias_value":"VITLZQIEI6MJTI5P","created_at":"2026-07-05T09:10:15.844577+00:00"},{"alias_kind":"pith_short_8","alias_value":"VITLZQIE","created_at":"2026-07-05T09:10:15.844577+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01232","citing_title":"Is One Layer Enough? Training A Single Transformer Layer Can Match Full-Parameter RL Training","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01232","citing_title":"Is One Layer Enough? Training A Single Transformer Layer Can Match Full-Parameter RL Training","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2505.17138","citing_title":"RAP: Runtime Adaptive Pruning for LLM Inference","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2511.06516","citing_title":"You Had One Job: Per-Task Quantization Using LLMs' Hidden Representations","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02404","citing_title":"Statistically-Lossless Quantization of Large Language Models","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19974","citing_title":"Are LLM Uncertainty and Correctness Encoded by the Same Features? A Functional Dissociation via Sparse Autoencoders","ref_index":55,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VITLZQIEI6MJTI5PTBL2DQA26E","json":"https://pith.science/pith/VITLZQIEI6MJTI5PTBL2DQA26E.json","graph_json":"https://pith.science/api/pith-number/VITLZQIEI6MJTI5PTBL2DQA26E/graph.json","events_json":"https://pith.science/api/pith-number/VITLZQIEI6MJTI5PTBL2DQA26E/events.json","paper":"https://pith.science/paper/VITLZQIE"},"agent_actions":{"view_html":"https://pith.science/pith/VITLZQIEI6MJTI5PTBL2DQA26E","download_json":"https://pith.science/pith/VITLZQIEI6MJTI5PTBL2DQA26E.json","view_paper":"https://pith.science/paper/VITLZQIE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.14381&json=true","fetch_graph":"https://pith.science/api/pith-number/VITLZQIEI6MJTI5PTBL2DQA26E/graph.json","fetch_events":"https://pith.science/api/pith-number/VITLZQIEI6MJTI5PTBL2DQA26E/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VITLZQIEI6MJTI5PTBL2DQA26E/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VITLZQIEI6MJTI5PTBL2DQA26E/action/storage_attestation","attest_author":"https://pith.science/pith/VITLZQIEI6MJTI5PTBL2DQA26E/action/author_attestation","sign_citation":"https://pith.science/pith/VITLZQIEI6MJTI5PTBL2DQA26E/action/citation_signature","submit_replication":"https://pith.science/pith/VITLZQIEI6MJTI5PTBL2DQA26E/action/replication_record"}},"created_at":"2026-07-05T09:10:15.844577+00:00","updated_at":"2026-07-05T09:10:15.844577+00:00"}