{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:DFAUMTN75Z35DB3KO3VR52M27M","short_pith_number":"pith:DFAUMTN7","schema_version":"1.0","canonical_sha256":"1941464dbfee77d1876a76eb1ee99afb1d4a3effbbc2822dc7405d4e7f2ddb02","source":{"kind":"arxiv","id":"2203.10705","version":2},"attestation_state":"computed","paper":{"title":"Compression of Generative Pre-trained Language Models via Quantization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Chaofan Tao, Lifeng Shang, Lu Hou, Ngai Wong, Ping Luo, Qun Liu, Wei Zhang, Xin Jiang","submitted_at":"2022-03-21T02:11:35Z","abstract_excerpt":"The increasing size of generative Pre-trained Language Models (PLMs) has greatly increased the demand for model compression. Despite various methods to compress BERT or its variants, there are few attempts to compress generative PLMs, and the underlying difficulty remains unclear. In this paper, we compress generative PLMs by quantization. We find that previous quantization methods fail on generative tasks due to the \\textit{homogeneous word embeddings} caused by reduced capacity, and \\textit{varied distribution of weights}. Correspondingly, we propose a token-level contrastive distillation to"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2203.10705","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2022-03-21T02:11:35Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"f76e7a891e8db2aaf6313c9511b6c3005e636b8f8507ea483e4c82919b637d6e","abstract_canon_sha256":"6b4f725c49c62a10927b31b1820aea3a853f38a683a7447eddfe811775237a05"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:40:52.567514Z","signature_b64":"0XR4IdHTZV2xAEhLFGyf/4dFt4vZ2957LnmjrgCf+/TYPEW1khqwafWjW6zl4ae1A3RTHWIY/COinGz5kH5rBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1941464dbfee77d1876a76eb1ee99afb1d4a3effbbc2822dc7405d4e7f2ddb02","last_reissued_at":"2026-07-05T04:40:52.567083Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:40:52.567083Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Compression of Generative Pre-trained Language Models via Quantization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Chaofan Tao, Lifeng Shang, Lu Hou, Ngai Wong, Ping Luo, Qun Liu, Wei Zhang, Xin Jiang","submitted_at":"2022-03-21T02:11:35Z","abstract_excerpt":"The increasing size of generative Pre-trained Language Models (PLMs) has greatly increased the demand for model compression. Despite various methods to compress BERT or its variants, there are few attempts to compress generative PLMs, and the underlying difficulty remains unclear. In this paper, we compress generative PLMs by quantization. We find that previous quantization methods fail on generative tasks due to the \\textit{homogeneous word embeddings} caused by reduced capacity, and \\textit{varied distribution of weights}. Correspondingly, we propose a token-level contrastive distillation to"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2203.10705","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2203.10705/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2203.10705","created_at":"2026-07-05T04:40:52.567139+00:00"},{"alias_kind":"arxiv_version","alias_value":"2203.10705v2","created_at":"2026-07-05T04:40:52.567139+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2203.10705","created_at":"2026-07-05T04:40:52.567139+00:00"},{"alias_kind":"pith_short_12","alias_value":"DFAUMTN75Z35","created_at":"2026-07-05T04:40:52.567139+00:00"},{"alias_kind":"pith_short_16","alias_value":"DFAUMTN75Z35DB3K","created_at":"2026-07-05T04:40:52.567139+00:00"},{"alias_kind":"pith_short_8","alias_value":"DFAUMTN7","created_at":"2026-07-05T04:40:52.567139+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2307.06435","citing_title":"A Comprehensive Overview of Large Language Models","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2402.13116","citing_title":"A Survey on Knowledge Distillation of Large Language Models","ref_index":20,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DFAUMTN75Z35DB3KO3VR52M27M","json":"https://pith.science/pith/DFAUMTN75Z35DB3KO3VR52M27M.json","graph_json":"https://pith.science/api/pith-number/DFAUMTN75Z35DB3KO3VR52M27M/graph.json","events_json":"https://pith.science/api/pith-number/DFAUMTN75Z35DB3KO3VR52M27M/events.json","paper":"https://pith.science/paper/DFAUMTN7"},"agent_actions":{"view_html":"https://pith.science/pith/DFAUMTN75Z35DB3KO3VR52M27M","download_json":"https://pith.science/pith/DFAUMTN75Z35DB3KO3VR52M27M.json","view_paper":"https://pith.science/paper/DFAUMTN7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2203.10705&json=true","fetch_graph":"https://pith.science/api/pith-number/DFAUMTN75Z35DB3KO3VR52M27M/graph.json","fetch_events":"https://pith.science/api/pith-number/DFAUMTN75Z35DB3KO3VR52M27M/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DFAUMTN75Z35DB3KO3VR52M27M/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DFAUMTN75Z35DB3KO3VR52M27M/action/storage_attestation","attest_author":"https://pith.science/pith/DFAUMTN75Z35DB3KO3VR52M27M/action/author_attestation","sign_citation":"https://pith.science/pith/DFAUMTN75Z35DB3KO3VR52M27M/action/citation_signature","submit_replication":"https://pith.science/pith/DFAUMTN75Z35DB3KO3VR52M27M/action/replication_record"}},"created_at":"2026-07-05T04:40:52.567139+00:00","updated_at":"2026-07-05T04:40:52.567139+00:00"}