{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:BNA7GLS242WQACXGSA5HHUYCRY","short_pith_number":"pith:BNA7GLS2","schema_version":"1.0","canonical_sha256":"0b41f32e5ae6ad000ae6903a73d3028e1bb6fa4df9ad8c9475b2b85de25fafab","source":{"kind":"arxiv","id":"2310.08041","version":3},"attestation_state":"computed","paper":{"title":"QLLM: Accurate and Efficient Low-Bitwidth Quantization for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Bohan Zhuang, Jianfei Cai, Jing Liu, Ruihao Gong, Xiuying Wei, Zhiwei Dong","submitted_at":"2023-10-12T05:25:49Z","abstract_excerpt":"Large Language Models (LLMs) excel in NLP, but their demands hinder their widespread deployment. While Quantization-Aware Training (QAT) offers a solution, its extensive training costs make Post-Training Quantization (PTQ) a more practical approach for LLMs. In existing studies, activation outliers in particular channels are identified as the bottleneck to PTQ accuracy. They propose to transform the magnitudes from activations to weights, which however offers limited alleviation or suffers from unstable gradients, resulting in a severe performance drop at low-bitwidth. In this paper, we propos"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.08041","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-10-12T05:25:49Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"31d63b7b72f895359ac96fb58aadc3adae3dbd292f3577927d59410c5563e848","abstract_canon_sha256":"2b1d529abf177f0c4b70e5377b257bdff82fc5b087c05367bd8a2c7790985254"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:05:03.747657Z","signature_b64":"00+bx8B3x9L/P+yw4gTDVGxKvt5TNy0qdxN2jZfkWTfwNAgaEqaBrgMB4+jnq9g2AugadeunUfOMUQx8frDJDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0b41f32e5ae6ad000ae6903a73d3028e1bb6fa4df9ad8c9475b2b85de25fafab","last_reissued_at":"2026-07-05T08:05:03.746941Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:05:03.746941Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"QLLM: Accurate and Efficient Low-Bitwidth Quantization for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Bohan Zhuang, Jianfei Cai, Jing Liu, Ruihao Gong, Xiuying Wei, Zhiwei Dong","submitted_at":"2023-10-12T05:25:49Z","abstract_excerpt":"Large Language Models (LLMs) excel in NLP, but their demands hinder their widespread deployment. While Quantization-Aware Training (QAT) offers a solution, its extensive training costs make Post-Training Quantization (PTQ) a more practical approach for LLMs. In existing studies, activation outliers in particular channels are identified as the bottleneck to PTQ accuracy. They propose to transform the magnitudes from activations to weights, which however offers limited alleviation or suffers from unstable gradients, resulting in a severe performance drop at low-bitwidth. In this paper, we propos"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.08041","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.08041/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.08041","created_at":"2026-07-05T08:05:03.747032+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.08041v3","created_at":"2026-07-05T08:05:03.747032+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.08041","created_at":"2026-07-05T08:05:03.747032+00:00"},{"alias_kind":"pith_short_12","alias_value":"BNA7GLS242WQ","created_at":"2026-07-05T08:05:03.747032+00:00"},{"alias_kind":"pith_short_16","alias_value":"BNA7GLS242WQACXG","created_at":"2026-07-05T08:05:03.747032+00:00"},{"alias_kind":"pith_short_8","alias_value":"BNA7GLS2","created_at":"2026-07-05T08:05:03.747032+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.00206","citing_title":"Quantized Reasoning Models Think They Need to Think Longer, but They Do Not","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2405.16406","citing_title":"SpinQuant: LLM quantization with learned rotations","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13907","citing_title":"AIS: Adaptive Importance Sampling for Quantized RL","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BNA7GLS242WQACXGSA5HHUYCRY","json":"https://pith.science/pith/BNA7GLS242WQACXGSA5HHUYCRY.json","graph_json":"https://pith.science/api/pith-number/BNA7GLS242WQACXGSA5HHUYCRY/graph.json","events_json":"https://pith.science/api/pith-number/BNA7GLS242WQACXGSA5HHUYCRY/events.json","paper":"https://pith.science/paper/BNA7GLS2"},"agent_actions":{"view_html":"https://pith.science/pith/BNA7GLS242WQACXGSA5HHUYCRY","download_json":"https://pith.science/pith/BNA7GLS242WQACXGSA5HHUYCRY.json","view_paper":"https://pith.science/paper/BNA7GLS2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.08041&json=true","fetch_graph":"https://pith.science/api/pith-number/BNA7GLS242WQACXGSA5HHUYCRY/graph.json","fetch_events":"https://pith.science/api/pith-number/BNA7GLS242WQACXGSA5HHUYCRY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BNA7GLS242WQACXGSA5HHUYCRY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BNA7GLS242WQACXGSA5HHUYCRY/action/storage_attestation","attest_author":"https://pith.science/pith/BNA7GLS242WQACXGSA5HHUYCRY/action/author_attestation","sign_citation":"https://pith.science/pith/BNA7GLS242WQACXGSA5HHUYCRY/action/citation_signature","submit_replication":"https://pith.science/pith/BNA7GLS242WQACXGSA5HHUYCRY/action/replication_record"}},"created_at":"2026-07-05T08:05:03.747032+00:00","updated_at":"2026-07-05T08:05:03.747032+00:00"}