{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:6QP4KWFGW6QYX2N7DSOXFEBE5A","short_pith_number":"pith:6QP4KWFG","schema_version":"1.0","canonical_sha256":"f41fc558a6b7a18be9bf1c9d729024e822ca7e1575d5162c460229f572ffecb9","source":{"kind":"arxiv","id":"2402.05964","version":2},"attestation_state":"computed","paper":{"title":"A Survey on Transformer Compression","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.CV"],"primary_cat":"cs.LG","authors_text":"Dacheng Tao, Hailin Hu, Jianyuan Guo, Kai Han, Yehui Tang, Yunhe Wang, Zhijun Tu","submitted_at":"2024-02-05T12:16:28Z","abstract_excerpt":"Transformer plays a vital role in the realms of natural language processing (NLP) and computer vision (CV), specially for constructing large language models (LLM) and large vision models (LVM). Model compression methods reduce the memory and computational cost of Transformer, which is a necessary step to implement large language/vision models on practical devices. Given the unique architecture of Transformer, featuring alternative attention and feedforward neural network (FFN) modules, specific compression techniques are usually required. The efficiency of these compression methods is also par"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.05964","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-02-05T12:16:28Z","cross_cats_sorted":["cs.CL","cs.CV"],"title_canon_sha256":"307cd7834cf3a1358315dbf1362d2c790dbc27b614dfaa385f4a1d37d4f3af43","abstract_canon_sha256":"9583657091d36356879842fdb97741238ec29fc293d7d9c546aefaf4ecd93def"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:05:10.287973Z","signature_b64":"60RMDJZLL0Zy2l9AK2wPpoBYQDWByiwyEhMZ/kjyPv9j773RlXNlJS/4irk+cV3a1T3IP+fJnoRICCBypTQ2BA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f41fc558a6b7a18be9bf1c9d729024e822ca7e1575d5162c460229f572ffecb9","last_reissued_at":"2026-07-05T08:05:10.287478Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:05:10.287478Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Survey on Transformer Compression","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.CV"],"primary_cat":"cs.LG","authors_text":"Dacheng Tao, Hailin Hu, Jianyuan Guo, Kai Han, Yehui Tang, Yunhe Wang, Zhijun Tu","submitted_at":"2024-02-05T12:16:28Z","abstract_excerpt":"Transformer plays a vital role in the realms of natural language processing (NLP) and computer vision (CV), specially for constructing large language models (LLM) and large vision models (LVM). Model compression methods reduce the memory and computational cost of Transformer, which is a necessary step to implement large language/vision models on practical devices. Given the unique architecture of Transformer, featuring alternative attention and feedforward neural network (FFN) modules, specific compression techniques are usually required. The efficiency of these compression methods is also par"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.05964","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.05964/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.05964","created_at":"2026-07-05T08:05:10.287540+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.05964v2","created_at":"2026-07-05T08:05:10.287540+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.05964","created_at":"2026-07-05T08:05:10.287540+00:00"},{"alias_kind":"pith_short_12","alias_value":"6QP4KWFGW6QY","created_at":"2026-07-05T08:05:10.287540+00:00"},{"alias_kind":"pith_short_16","alias_value":"6QP4KWFGW6QYX2N7","created_at":"2026-07-05T08:05:10.287540+00:00"},{"alias_kind":"pith_short_8","alias_value":"6QP4KWFG","created_at":"2026-07-05T08:05:10.287540+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.04032","citing_title":"Do Transformers Need Three Projections? Systematic Study of QKV Variants","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2502.02970","citing_title":"Distributional Statistics Restore Training Data Auditability in One-step Distilled Diffusion Models","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2508.06974","citing_title":"Rethinking 1-bit Optimization Leveraging Pre-trained Large Language Models","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2602.06252","citing_title":"D-Legion: A Scalable Many-Core Architecture for Accelerating Matrix Multiplication in Quantized LLMs","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2404.14294","citing_title":"A Survey on Efficient Inference for Large Language Models","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17286","citing_title":"Depth Adaptive Efficient Visual Autoregressive Modeling","ref_index":56,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6QP4KWFGW6QYX2N7DSOXFEBE5A","json":"https://pith.science/pith/6QP4KWFGW6QYX2N7DSOXFEBE5A.json","graph_json":"https://pith.science/api/pith-number/6QP4KWFGW6QYX2N7DSOXFEBE5A/graph.json","events_json":"https://pith.science/api/pith-number/6QP4KWFGW6QYX2N7DSOXFEBE5A/events.json","paper":"https://pith.science/paper/6QP4KWFG"},"agent_actions":{"view_html":"https://pith.science/pith/6QP4KWFGW6QYX2N7DSOXFEBE5A","download_json":"https://pith.science/pith/6QP4KWFGW6QYX2N7DSOXFEBE5A.json","view_paper":"https://pith.science/paper/6QP4KWFG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.05964&json=true","fetch_graph":"https://pith.science/api/pith-number/6QP4KWFGW6QYX2N7DSOXFEBE5A/graph.json","fetch_events":"https://pith.science/api/pith-number/6QP4KWFGW6QYX2N7DSOXFEBE5A/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6QP4KWFGW6QYX2N7DSOXFEBE5A/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6QP4KWFGW6QYX2N7DSOXFEBE5A/action/storage_attestation","attest_author":"https://pith.science/pith/6QP4KWFGW6QYX2N7DSOXFEBE5A/action/author_attestation","sign_citation":"https://pith.science/pith/6QP4KWFGW6QYX2N7DSOXFEBE5A/action/citation_signature","submit_replication":"https://pith.science/pith/6QP4KWFGW6QYX2N7DSOXFEBE5A/action/replication_record"}},"created_at":"2026-07-05T08:05:10.287540+00:00","updated_at":"2026-07-05T08:05:10.287540+00:00"}