{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:7AUTBKT2K2NK2GX44W6X22NVYC","short_pith_number":"pith:7AUTBKT2","schema_version":"1.0","canonical_sha256":"f82930aa7a569aad1afce5bd7d69b5c0ab60a278806dea09ee6a04c1eb123103","source":{"kind":"arxiv","id":"2402.16775","version":2},"attestation_state":"computed","paper":{"title":"A Comprehensive Evaluation of Quantization Strategies for Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Bin Wang, Deyi Xiong, Jiangcun Du, Jian Luan, Renren Jin, Wei Liu, Wuwei Huang","submitted_at":"2024-02-26T17:45:36Z","abstract_excerpt":"Increasing the number of parameters in large language models (LLMs) usually improves performance in downstream tasks but raises compute and memory costs, making deployment difficult in resource-limited settings. Quantization techniques, which reduce the bits needed for model weights or activations with minimal performance loss, have become popular due to the rise of LLMs. However, most quantization studies use pre-trained LLMs, and the impact of quantization on instruction-tuned LLMs and the relationship between perplexity and benchmark performance of quantized LLMs are not well understood. Ev"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.16775","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-02-26T17:45:36Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"2b74e3d5ceab53170d620c659e4154bee75217ad900b38d47e6b797bfc3cc3b1","abstract_canon_sha256":"a52f5cb8cff0832551d5093c0d4cfab144985783cd8f0accde96b7b4db23c4f8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:28:07.039896Z","signature_b64":"zNZ1jno6MKVG0d/H+qKGuiskSlnXuVAD6YcnoK5If6Ou9HNKrP85jVUPo10EdIXB8raM9KMpokNFELrNVi1/Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f82930aa7a569aad1afce5bd7d69b5c0ab60a278806dea09ee6a04c1eb123103","last_reissued_at":"2026-07-05T08:28:07.039403Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:28:07.039403Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Comprehensive Evaluation of Quantization Strategies for Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Bin Wang, Deyi Xiong, Jiangcun Du, Jian Luan, Renren Jin, Wei Liu, Wuwei Huang","submitted_at":"2024-02-26T17:45:36Z","abstract_excerpt":"Increasing the number of parameters in large language models (LLMs) usually improves performance in downstream tasks but raises compute and memory costs, making deployment difficult in resource-limited settings. Quantization techniques, which reduce the bits needed for model weights or activations with minimal performance loss, have become popular due to the rise of LLMs. However, most quantization studies use pre-trained LLMs, and the impact of quantization on instruction-tuned LLMs and the relationship between perplexity and benchmark performance of quantized LLMs are not well understood. Ev"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.16775","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.16775/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.16775","created_at":"2026-07-05T08:28:07.039462+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.16775v2","created_at":"2026-07-05T08:28:07.039462+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.16775","created_at":"2026-07-05T08:28:07.039462+00:00"},{"alias_kind":"pith_short_12","alias_value":"7AUTBKT2K2NK","created_at":"2026-07-05T08:28:07.039462+00:00"},{"alias_kind":"pith_short_16","alias_value":"7AUTBKT2K2NK2GX4","created_at":"2026-07-05T08:28:07.039462+00:00"},{"alias_kind":"pith_short_8","alias_value":"7AUTBKT2","created_at":"2026-07-05T08:28:07.039462+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.19884","citing_title":"From Signal Degradation to Computation Collapse: Uncovering the Two Failure Modes of LLM Quantization","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7AUTBKT2K2NK2GX44W6X22NVYC","json":"https://pith.science/pith/7AUTBKT2K2NK2GX44W6X22NVYC.json","graph_json":"https://pith.science/api/pith-number/7AUTBKT2K2NK2GX44W6X22NVYC/graph.json","events_json":"https://pith.science/api/pith-number/7AUTBKT2K2NK2GX44W6X22NVYC/events.json","paper":"https://pith.science/paper/7AUTBKT2"},"agent_actions":{"view_html":"https://pith.science/pith/7AUTBKT2K2NK2GX44W6X22NVYC","download_json":"https://pith.science/pith/7AUTBKT2K2NK2GX44W6X22NVYC.json","view_paper":"https://pith.science/paper/7AUTBKT2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.16775&json=true","fetch_graph":"https://pith.science/api/pith-number/7AUTBKT2K2NK2GX44W6X22NVYC/graph.json","fetch_events":"https://pith.science/api/pith-number/7AUTBKT2K2NK2GX44W6X22NVYC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7AUTBKT2K2NK2GX44W6X22NVYC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7AUTBKT2K2NK2GX44W6X22NVYC/action/storage_attestation","attest_author":"https://pith.science/pith/7AUTBKT2K2NK2GX44W6X22NVYC/action/author_attestation","sign_citation":"https://pith.science/pith/7AUTBKT2K2NK2GX44W6X22NVYC/action/citation_signature","submit_replication":"https://pith.science/pith/7AUTBKT2K2NK2GX44W6X22NVYC/action/replication_record"}},"created_at":"2026-07-05T08:28:07.039462+00:00","updated_at":"2026-07-05T08:28:07.039462+00:00"}