{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:H2GLEEKLAALOYJUHPTATIQVKWC","short_pith_number":"pith:H2GLEEKL","schema_version":"1.0","canonical_sha256":"3e8cb2114b0016ec26877cc13442aab090538a575567d96e1302e7331770d691","source":{"kind":"arxiv","id":"2407.16286","version":1},"attestation_state":"computed","paper":{"title":"A deeper look at depth pruning of LLMs","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"David Krueger, Greg Heinrich, Jan Kautz, Pavlo Molchanov, Shoaib Ahmed Siddiqui, Thomas Breuel, Xin Dong","submitted_at":"2024-07-23T08:40:27Z","abstract_excerpt":"Large Language Models (LLMs) are not only resource-intensive to train but even more costly to deploy in production. Therefore, recent work has attempted to prune blocks of LLMs based on cheap proxies for estimating block importance, effectively removing 10% of blocks in well-trained LLaMa-2 and Mistral 7b models without any significant degradation of downstream metrics. In this paper, we explore different block importance metrics by considering adaptive metrics such as Shapley value in addition to static ones explored in prior work. We show that adaptive metrics exhibit a trade-off in performa"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.16286","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2024-07-23T08:40:27Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"caa1ceb81043c799a8acf1897b6418602c5d5da4007879f9e518307eb3e14515","abstract_canon_sha256":"6e6e12906f6a2a81e83c5d5aec810068ba9fa3c474a3c0fadcb631b440e82311"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:47:30.848541Z","signature_b64":"AuJ8H+AJNp1MBHKshnU+FYkcfwHHE6sXzJc80bgSr9uninKSct4PMq+DNYhp+WDkMBA6QYzyFGmETMn4VPnlAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3e8cb2114b0016ec26877cc13442aab090538a575567d96e1302e7331770d691","last_reissued_at":"2026-07-05T08:47:30.848131Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:47:30.848131Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A deeper look at depth pruning of LLMs","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"David Krueger, Greg Heinrich, Jan Kautz, Pavlo Molchanov, Shoaib Ahmed Siddiqui, Thomas Breuel, Xin Dong","submitted_at":"2024-07-23T08:40:27Z","abstract_excerpt":"Large Language Models (LLMs) are not only resource-intensive to train but even more costly to deploy in production. Therefore, recent work has attempted to prune blocks of LLMs based on cheap proxies for estimating block importance, effectively removing 10% of blocks in well-trained LLaMa-2 and Mistral 7b models without any significant degradation of downstream metrics. In this paper, we explore different block importance metrics by considering adaptive metrics such as Shapley value in addition to static ones explored in prior work. We show that adaptive metrics exhibit a trade-off in performa"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.16286","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.16286/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.16286","created_at":"2026-07-05T08:47:30.848191+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.16286v1","created_at":"2026-07-05T08:47:30.848191+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.16286","created_at":"2026-07-05T08:47:30.848191+00:00"},{"alias_kind":"pith_short_12","alias_value":"H2GLEEKLAALO","created_at":"2026-07-05T08:47:30.848191+00:00"},{"alias_kind":"pith_short_16","alias_value":"H2GLEEKLAALOYJUH","created_at":"2026-07-05T08:47:30.848191+00:00"},{"alias_kind":"pith_short_8","alias_value":"H2GLEEKL","created_at":"2026-07-05T08:47:30.848191+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2602.01167","citing_title":"Do All Individual Layers Help? An Empirical Study of Task-Interfering Layers in Vision-Language Models","ref_index":40,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/H2GLEEKLAALOYJUHPTATIQVKWC","json":"https://pith.science/pith/H2GLEEKLAALOYJUHPTATIQVKWC.json","graph_json":"https://pith.science/api/pith-number/H2GLEEKLAALOYJUHPTATIQVKWC/graph.json","events_json":"https://pith.science/api/pith-number/H2GLEEKLAALOYJUHPTATIQVKWC/events.json","paper":"https://pith.science/paper/H2GLEEKL"},"agent_actions":{"view_html":"https://pith.science/pith/H2GLEEKLAALOYJUHPTATIQVKWC","download_json":"https://pith.science/pith/H2GLEEKLAALOYJUHPTATIQVKWC.json","view_paper":"https://pith.science/paper/H2GLEEKL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.16286&json=true","fetch_graph":"https://pith.science/api/pith-number/H2GLEEKLAALOYJUHPTATIQVKWC/graph.json","fetch_events":"https://pith.science/api/pith-number/H2GLEEKLAALOYJUHPTATIQVKWC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/H2GLEEKLAALOYJUHPTATIQVKWC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/H2GLEEKLAALOYJUHPTATIQVKWC/action/storage_attestation","attest_author":"https://pith.science/pith/H2GLEEKLAALOYJUHPTATIQVKWC/action/author_attestation","sign_citation":"https://pith.science/pith/H2GLEEKLAALOYJUHPTATIQVKWC/action/citation_signature","submit_replication":"https://pith.science/pith/H2GLEEKLAALOYJUHPTATIQVKWC/action/replication_record"}},"created_at":"2026-07-05T08:47:30.848191+00:00","updated_at":"2026-07-05T08:47:30.848191+00:00"}