{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:G6NACNUNUD4BMEYC77DTEYXLDV","short_pith_number":"pith:G6NACNUN","schema_version":"1.0","canonical_sha256":"379a01368da0f8161302ffc73262eb1d4834b0bae61ce464ace32e8374cba20b","source":{"kind":"arxiv","id":"2109.04838","version":1},"attestation_state":"computed","paper":{"title":"Block Pruning For Faster Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Alexander M. Rush, Ella Charlaix, Fran\\c{c}ois Lagunas, Victor Sanh","submitted_at":"2021-09-10T12:46:32Z","abstract_excerpt":"Pre-training has improved model accuracy for both classification and generation tasks at the cost of introducing much larger and slower models. Pruning methods have proven to be an effective way of reducing model size, whereas distillation methods are proven for speeding up inference. We introduce a block pruning approach targeting both small and fast models. Our approach extends structured methods by considering blocks of any size and integrates this structure into the movement pruning paradigm for fine-tuning. We find that this approach learns to prune out full components of the underlying m"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2109.04838","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-09-10T12:46:32Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"8896fb5116657a9b1d327c50b8894e40cdaaec4ad8f4813cd026444081267f84","abstract_canon_sha256":"d9047beb0e470c99f7b99070782a73ee6f1e2e653b95d78980b3fe65eb6e9c69"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:13:10.658792Z","signature_b64":"vU+vHY0mBpP1M5TKnKwqZ5WL4HMYQT/FbYNngTi6mknymlywrfK+CBwPTFm3eGGGSNMwb6tVytpGlVvLt0NRDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"379a01368da0f8161302ffc73262eb1d4834b0bae61ce464ace32e8374cba20b","last_reissued_at":"2026-07-05T03:13:10.658371Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:13:10.658371Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Block Pruning For Faster Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Alexander M. Rush, Ella Charlaix, Fran\\c{c}ois Lagunas, Victor Sanh","submitted_at":"2021-09-10T12:46:32Z","abstract_excerpt":"Pre-training has improved model accuracy for both classification and generation tasks at the cost of introducing much larger and slower models. Pruning methods have proven to be an effective way of reducing model size, whereas distillation methods are proven for speeding up inference. We introduce a block pruning approach targeting both small and fast models. Our approach extends structured methods by considering blocks of any size and integrates this structure into the movement pruning paradigm for fine-tuning. We find that this approach learns to prune out full components of the underlying m"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2109.04838","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2109.04838/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2109.04838","created_at":"2026-07-05T03:13:10.658429+00:00"},{"alias_kind":"arxiv_version","alias_value":"2109.04838v1","created_at":"2026-07-05T03:13:10.658429+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2109.04838","created_at":"2026-07-05T03:13:10.658429+00:00"},{"alias_kind":"pith_short_12","alias_value":"G6NACNUNUD4B","created_at":"2026-07-05T03:13:10.658429+00:00"},{"alias_kind":"pith_short_16","alias_value":"G6NACNUNUD4BMEYC","created_at":"2026-07-05T03:13:10.658429+00:00"},{"alias_kind":"pith_short_8","alias_value":"G6NACNUN","created_at":"2026-07-05T03:13:10.658429+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2310.02277","citing_title":"Junk DNA Hypothesis: Pruning Small Pre-Trained Weights Irreversibly and Monotonically Impairs \"Difficult\" Downstream Tasks in LLMs","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14738","citing_title":"TAPIOCA: Why Task- Aware Pruning Improves OOD model Capability","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22132","citing_title":"Accelerating Vision Foundation Models with Drop-in Depthwise Convolution","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2510.22767","citing_title":"TELL-TALE: Task Efficient LLMs with Task Aware Layer Elimination","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/G6NACNUNUD4BMEYC77DTEYXLDV","json":"https://pith.science/pith/G6NACNUNUD4BMEYC77DTEYXLDV.json","graph_json":"https://pith.science/api/pith-number/G6NACNUNUD4BMEYC77DTEYXLDV/graph.json","events_json":"https://pith.science/api/pith-number/G6NACNUNUD4BMEYC77DTEYXLDV/events.json","paper":"https://pith.science/paper/G6NACNUN"},"agent_actions":{"view_html":"https://pith.science/pith/G6NACNUNUD4BMEYC77DTEYXLDV","download_json":"https://pith.science/pith/G6NACNUNUD4BMEYC77DTEYXLDV.json","view_paper":"https://pith.science/paper/G6NACNUN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2109.04838&json=true","fetch_graph":"https://pith.science/api/pith-number/G6NACNUNUD4BMEYC77DTEYXLDV/graph.json","fetch_events":"https://pith.science/api/pith-number/G6NACNUNUD4BMEYC77DTEYXLDV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/G6NACNUNUD4BMEYC77DTEYXLDV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/G6NACNUNUD4BMEYC77DTEYXLDV/action/storage_attestation","attest_author":"https://pith.science/pith/G6NACNUNUD4BMEYC77DTEYXLDV/action/author_attestation","sign_citation":"https://pith.science/pith/G6NACNUNUD4BMEYC77DTEYXLDV/action/citation_signature","submit_replication":"https://pith.science/pith/G6NACNUNUD4BMEYC77DTEYXLDV/action/replication_record"}},"created_at":"2026-07-05T03:13:10.658429+00:00","updated_at":"2026-07-05T03:13:10.658429+00:00"}