{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:5SZUXPGZJI63HB7YZI35UP6D57","short_pith_number":"pith:5SZUXPGZ","schema_version":"1.0","canonical_sha256":"ecb34bbcd94a3db387f8ca37da3fc3efd964ac5bdaaa9b090d0779424f3a745d","source":{"kind":"arxiv","id":"2307.13304","version":2},"attestation_state":"computed","paper":{"title":"QuIP: 2-Bit Quantization of Large Language Models With Guarantees","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Christopher De Sa, Jerry Chee, Volodymyr Kuleshov, Yaohui Cai","submitted_at":"2023-07-25T07:44:06Z","abstract_excerpt":"This work studies post-training parameter quantization in large language models (LLMs). We introduce quantization with incoherence processing (QuIP), a new method based on the insight that quantization benefits from $\\textit{incoherent}$ weight and Hessian matrices, i.e., from the weights being even in magnitude and the directions in which it is important to round them accurately being unaligned with the coordinate axes. QuIP consists of two steps: (1) an adaptive rounding procedure minimizing a quadratic proxy objective; (2) efficient pre- and post-processing that ensures weight and Hessian i"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.13304","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-07-25T07:44:06Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"11b2edb88e85c56877edfd5365cdbbd69232687243917ebdb6898cee9dd91a56","abstract_canon_sha256":"dec98e06af9bcb5f9894a79c539e2c086c77aa2f0fa6cbf10e494d6ed7a5bd3f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:33:53.415765Z","signature_b64":"Aiwbl7FRowcylv49TxzxYZwIsrqva9cyE9bSVReJEuMtXKAQUxb5sgXzQlFxZRUlx7GwaqIFE0lq/M8vzq3zAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ecb34bbcd94a3db387f8ca37da3fc3efd964ac5bdaaa9b090d0779424f3a745d","last_reissued_at":"2026-07-05T07:33:53.415221Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:33:53.415221Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"QuIP: 2-Bit Quantization of Large Language Models With Guarantees","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Christopher De Sa, Jerry Chee, Volodymyr Kuleshov, Yaohui Cai","submitted_at":"2023-07-25T07:44:06Z","abstract_excerpt":"This work studies post-training parameter quantization in large language models (LLMs). We introduce quantization with incoherence processing (QuIP), a new method based on the insight that quantization benefits from $\\textit{incoherent}$ weight and Hessian matrices, i.e., from the weights being even in magnitude and the directions in which it is important to round them accurately being unaligned with the coordinate axes. QuIP consists of two steps: (1) an adaptive rounding procedure minimizing a quadratic proxy objective; (2) efficient pre- and post-processing that ensures weight and Hessian i"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.13304","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.13304/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.13304","created_at":"2026-07-05T07:33:53.415283+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.13304v2","created_at":"2026-07-05T07:33:53.415283+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.13304","created_at":"2026-07-05T07:33:53.415283+00:00"},{"alias_kind":"pith_short_12","alias_value":"5SZUXPGZJI63","created_at":"2026-07-05T07:33:53.415283+00:00"},{"alias_kind":"pith_short_16","alias_value":"5SZUXPGZJI63HB7Y","created_at":"2026-07-05T07:33:53.415283+00:00"},{"alias_kind":"pith_short_8","alias_value":"5SZUXPGZ","created_at":"2026-07-05T07:33:53.415283+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23406","citing_title":"HyperQuant: A Rate-Distortion-Optimal Quantization Pipeline for Large Language and Diffusion Models","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14844","citing_title":"XFP: Quality-Targeted Adaptive Codebook Quantization with Sparse Outlier Separation for LLM Inference","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25203","citing_title":"Influence-Inspired Spectral Rotations for Extreme Low-Bit LLM Quantization","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2310.11453","citing_title":"BitNet: Scaling 1-bit Transformers for Large Language Models","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2402.17764","citing_title":"The Era of 1-bit LLMs: All Large Language Models are in 1.58 Bits","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2401.10774","citing_title":"Medusa: Simple LLM Inference Acceleration Framework with Multiple Decoding Heads","ref_index":78,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05699","citing_title":"When Quantization Is Free: An int4 KV Cache That Outruns fp16 on Apple Silicon","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24273","citing_title":"BitRL: Reinforcement Learning with 1-bit Quantized Language Models for Resource-Constrained Edge Deployment","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5SZUXPGZJI63HB7YZI35UP6D57","json":"https://pith.science/pith/5SZUXPGZJI63HB7YZI35UP6D57.json","graph_json":"https://pith.science/api/pith-number/5SZUXPGZJI63HB7YZI35UP6D57/graph.json","events_json":"https://pith.science/api/pith-number/5SZUXPGZJI63HB7YZI35UP6D57/events.json","paper":"https://pith.science/paper/5SZUXPGZ"},"agent_actions":{"view_html":"https://pith.science/pith/5SZUXPGZJI63HB7YZI35UP6D57","download_json":"https://pith.science/pith/5SZUXPGZJI63HB7YZI35UP6D57.json","view_paper":"https://pith.science/paper/5SZUXPGZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.13304&json=true","fetch_graph":"https://pith.science/api/pith-number/5SZUXPGZJI63HB7YZI35UP6D57/graph.json","fetch_events":"https://pith.science/api/pith-number/5SZUXPGZJI63HB7YZI35UP6D57/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5SZUXPGZJI63HB7YZI35UP6D57/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5SZUXPGZJI63HB7YZI35UP6D57/action/storage_attestation","attest_author":"https://pith.science/pith/5SZUXPGZJI63HB7YZI35UP6D57/action/author_attestation","sign_citation":"https://pith.science/pith/5SZUXPGZJI63HB7YZI35UP6D57/action/citation_signature","submit_replication":"https://pith.science/pith/5SZUXPGZJI63HB7YZI35UP6D57/action/replication_record"}},"created_at":"2026-07-05T07:33:53.415283+00:00","updated_at":"2026-07-05T07:33:53.415283+00:00"}