{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:FWRCPSCUPCYUY24RFAJKG7Y5AB","short_pith_number":"pith:FWRCPSCU","schema_version":"1.0","canonical_sha256":"2da227c85478b14c6b912812a37f1d0056617ae0425d010b61f62399645418d2","source":{"kind":"arxiv","id":"2305.17328","version":3},"attestation_state":"computed","paper":{"title":"Zero-TPrune: Zero-Shot Token Pruning through Leveraging of the Attention Graph in Pre-Trained Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","eess.IV"],"primary_cat":"cs.CV","authors_text":"Bhishma Dedhia, Hongjie Wang, Niraj K. Jha","submitted_at":"2023-05-27T02:08:51Z","abstract_excerpt":"Deployment of Transformer models on edge devices is becoming increasingly challenging due to the exponentially growing inference cost that scales quadratically with the number of tokens in the input sequence. Token pruning is an emerging solution to address this challenge due to its ease of deployment on various Transformer backbones. However, most token pruning methods require computationally expensive fine-tuning, which is undesirable in many edge deployment cases. In this work, we propose Zero-TPrune, the first zero-shot method that considers both the importance and similarity of tokens in "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.17328","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-05-27T02:08:51Z","cross_cats_sorted":["cs.AI","cs.LG","eess.IV"],"title_canon_sha256":"73ae9cf79f8bbba6a9fc636c95c59e73cd573828d748fda94e5a3043a05df0a0","abstract_canon_sha256":"2a0a5723b1757c57c32eb8149a661e90064caa22525b5c9dc2578106e7c9938a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:04:58.668638Z","signature_b64":"W7nCszHzawP4TzUxagwPpip+sVlcdZrEMsVAUo5pzMBCX7Xx98UfQyQCwL8c9HAMo+HpG+S7vCmF9eHZ9SGkBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2da227c85478b14c6b912812a37f1d0056617ae0425d010b61f62399645418d2","last_reissued_at":"2026-07-05T08:04:58.668130Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:04:58.668130Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Zero-TPrune: Zero-Shot Token Pruning through Leveraging of the Attention Graph in Pre-Trained Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","eess.IV"],"primary_cat":"cs.CV","authors_text":"Bhishma Dedhia, Hongjie Wang, Niraj K. Jha","submitted_at":"2023-05-27T02:08:51Z","abstract_excerpt":"Deployment of Transformer models on edge devices is becoming increasingly challenging due to the exponentially growing inference cost that scales quadratically with the number of tokens in the input sequence. Token pruning is an emerging solution to address this challenge due to its ease of deployment on various Transformer backbones. However, most token pruning methods require computationally expensive fine-tuning, which is undesirable in many edge deployment cases. In this work, we propose Zero-TPrune, the first zero-shot method that considers both the importance and similarity of tokens in "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.17328","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.17328/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.17328","created_at":"2026-07-05T08:04:58.668190+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.17328v3","created_at":"2026-07-05T08:04:58.668190+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.17328","created_at":"2026-07-05T08:04:58.668190+00:00"},{"alias_kind":"pith_short_12","alias_value":"FWRCPSCUPCYU","created_at":"2026-07-05T08:04:58.668190+00:00"},{"alias_kind":"pith_short_16","alias_value":"FWRCPSCUPCYUY24R","created_at":"2026-07-05T08:04:58.668190+00:00"},{"alias_kind":"pith_short_8","alias_value":"FWRCPSCU","created_at":"2026-07-05T08:04:58.668190+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2602.05243","citing_title":"CORP: Closed-Form One-shot Representation-Preserving Structured Pruning for Transformers","ref_index":20,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FWRCPSCUPCYUY24RFAJKG7Y5AB","json":"https://pith.science/pith/FWRCPSCUPCYUY24RFAJKG7Y5AB.json","graph_json":"https://pith.science/api/pith-number/FWRCPSCUPCYUY24RFAJKG7Y5AB/graph.json","events_json":"https://pith.science/api/pith-number/FWRCPSCUPCYUY24RFAJKG7Y5AB/events.json","paper":"https://pith.science/paper/FWRCPSCU"},"agent_actions":{"view_html":"https://pith.science/pith/FWRCPSCUPCYUY24RFAJKG7Y5AB","download_json":"https://pith.science/pith/FWRCPSCUPCYUY24RFAJKG7Y5AB.json","view_paper":"https://pith.science/paper/FWRCPSCU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.17328&json=true","fetch_graph":"https://pith.science/api/pith-number/FWRCPSCUPCYUY24RFAJKG7Y5AB/graph.json","fetch_events":"https://pith.science/api/pith-number/FWRCPSCUPCYUY24RFAJKG7Y5AB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FWRCPSCUPCYUY24RFAJKG7Y5AB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FWRCPSCUPCYUY24RFAJKG7Y5AB/action/storage_attestation","attest_author":"https://pith.science/pith/FWRCPSCUPCYUY24RFAJKG7Y5AB/action/author_attestation","sign_citation":"https://pith.science/pith/FWRCPSCUPCYUY24RFAJKG7Y5AB/action/citation_signature","submit_replication":"https://pith.science/pith/FWRCPSCUPCYUY24RFAJKG7Y5AB/action/replication_record"}},"created_at":"2026-07-05T08:04:58.668190+00:00","updated_at":"2026-07-05T08:04:58.668190+00:00"}