{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:6RIB7IE7ZG62L3HGROKOOKVXRJ","short_pith_number":"pith:6RIB7IE7","schema_version":"1.0","canonical_sha256":"f4501fa09fc9bda5ece68b94e72ab78a6779f359938ae72244b39eabf83e0a03","source":{"kind":"arxiv","id":"2504.03794","version":1},"attestation_state":"computed","paper":{"title":"Entropy-Based Block Pruning for Efficient Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Caiming Xiong, Doyen Sahoo, Huan Wang, Juntao Tan, Liangwei Yang, Shelby Heinecke, Silvio Savarese, Yuhui Xu","submitted_at":"2025-04-04T03:42:34Z","abstract_excerpt":"As large language models continue to scale, their growing computational and storage demands pose significant challenges for real-world deployment. In this work, we investigate redundancy within Transformer-based models and propose an entropy-based pruning strategy to enhance efficiency while maintaining performance. Empirical analysis reveals that the entropy of hidden representations decreases in the early blocks but progressively increases across most subsequent blocks. This trend suggests that entropy serves as a more effective measure of information richness within computation blocks. Unli"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.03794","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-04-04T03:42:34Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"28374dae14f397f3588f1ab2647def73d08f73fc39353e3b6a6a3420d6c27d7c","abstract_canon_sha256":"63ee40320674baf9629bc6a7a2ca1ee8d2107e299332c9684663d7e03f5d5be9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:44:48.543567Z","signature_b64":"knoFnE8vEZjD0Od7NWmZtoqhFaOyZsaUS1dNwLUOm13xjN+5Lxkxn+Heuk4U+s3QXzYiN27c2R7Fbv1crqjfDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f4501fa09fc9bda5ece68b94e72ab78a6779f359938ae72244b39eabf83e0a03","last_reissued_at":"2026-07-05T10:44:48.543038Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:44:48.543038Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Entropy-Based Block Pruning for Efficient Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Caiming Xiong, Doyen Sahoo, Huan Wang, Juntao Tan, Liangwei Yang, Shelby Heinecke, Silvio Savarese, Yuhui Xu","submitted_at":"2025-04-04T03:42:34Z","abstract_excerpt":"As large language models continue to scale, their growing computational and storage demands pose significant challenges for real-world deployment. In this work, we investigate redundancy within Transformer-based models and propose an entropy-based pruning strategy to enhance efficiency while maintaining performance. Empirical analysis reveals that the entropy of hidden representations decreases in the early blocks but progressively increases across most subsequent blocks. This trend suggests that entropy serves as a more effective measure of information richness within computation blocks. Unli"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.03794","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.03794/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.03794","created_at":"2026-07-05T10:44:48.543103+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.03794v1","created_at":"2026-07-05T10:44:48.543103+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.03794","created_at":"2026-07-05T10:44:48.543103+00:00"},{"alias_kind":"pith_short_12","alias_value":"6RIB7IE7ZG62","created_at":"2026-07-05T10:44:48.543103+00:00"},{"alias_kind":"pith_short_16","alias_value":"6RIB7IE7ZG62L3HG","created_at":"2026-07-05T10:44:48.543103+00:00"},{"alias_kind":"pith_short_8","alias_value":"6RIB7IE7","created_at":"2026-07-05T10:44:48.543103+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.19520","citing_title":"SimDiff: Depth Pruning via Similarity and Difference","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6RIB7IE7ZG62L3HGROKOOKVXRJ","json":"https://pith.science/pith/6RIB7IE7ZG62L3HGROKOOKVXRJ.json","graph_json":"https://pith.science/api/pith-number/6RIB7IE7ZG62L3HGROKOOKVXRJ/graph.json","events_json":"https://pith.science/api/pith-number/6RIB7IE7ZG62L3HGROKOOKVXRJ/events.json","paper":"https://pith.science/paper/6RIB7IE7"},"agent_actions":{"view_html":"https://pith.science/pith/6RIB7IE7ZG62L3HGROKOOKVXRJ","download_json":"https://pith.science/pith/6RIB7IE7ZG62L3HGROKOOKVXRJ.json","view_paper":"https://pith.science/paper/6RIB7IE7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.03794&json=true","fetch_graph":"https://pith.science/api/pith-number/6RIB7IE7ZG62L3HGROKOOKVXRJ/graph.json","fetch_events":"https://pith.science/api/pith-number/6RIB7IE7ZG62L3HGROKOOKVXRJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6RIB7IE7ZG62L3HGROKOOKVXRJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6RIB7IE7ZG62L3HGROKOOKVXRJ/action/storage_attestation","attest_author":"https://pith.science/pith/6RIB7IE7ZG62L3HGROKOOKVXRJ/action/author_attestation","sign_citation":"https://pith.science/pith/6RIB7IE7ZG62L3HGROKOOKVXRJ/action/citation_signature","submit_replication":"https://pith.science/pith/6RIB7IE7ZG62L3HGROKOOKVXRJ/action/replication_record"}},"created_at":"2026-07-05T10:44:48.543103+00:00","updated_at":"2026-07-05T10:44:48.543103+00:00"}