{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:23BIOGC3JHG6G3C4PIYQSVKVZB","short_pith_number":"pith:23BIOGC3","schema_version":"1.0","canonical_sha256":"d6c287185b49cde36c5c7a31095555c843cdebcefad27c9e9b7416c0e685ec49","source":{"kind":"arxiv","id":"2310.09259","version":2},"attestation_state":"computed","paper":{"title":"QUIK: Towards End-to-End 4-Bit Inference on Generative Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Dan Alistarh, Elias Frantar, Ilia Markov, Jie Ren, Saleh Ashkboos, Tingxuan Zhong, Torsten Hoefler, Xincheng Wang","submitted_at":"2023-10-13T17:15:05Z","abstract_excerpt":"Large Language Models (LLMs) from the GPT family have become extremely popular, leading to a race towards reducing their inference costs to allow for efficient local computation. Yet, the vast majority of existing work focuses on weight-only quantization, which can reduce runtime costs in the memory-bound one-token-at-a-time generative setting, but does not address them in compute-bound scenarios, such as batched inference or prompt processing. In this paper, we address the general quantization problem, where both weights and activations should be quantized. We show, for the first time, that t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.09259","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-10-13T17:15:05Z","cross_cats_sorted":[],"title_canon_sha256":"883fbca8398a7e2423199959d2d183ab7d48750907ed60fdd4ea3891f1969068","abstract_canon_sha256":"503ad0a6db8447feb44ea34fee8f062346324a1a3285800789e032ca70f4fc88"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:08:13.119544Z","signature_b64":"T6CbdmKiCIp8XS2vSqRo3FBuA64f1Krtqmg26iUn87700LC7TO3WIxgDZ6lbdcgkD1t55aqhTsXQWUSXVyoiAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d6c287185b49cde36c5c7a31095555c843cdebcefad27c9e9b7416c0e685ec49","last_reissued_at":"2026-07-05T07:08:13.118995Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:08:13.118995Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"QUIK: Towards End-to-End 4-Bit Inference on Generative Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Dan Alistarh, Elias Frantar, Ilia Markov, Jie Ren, Saleh Ashkboos, Tingxuan Zhong, Torsten Hoefler, Xincheng Wang","submitted_at":"2023-10-13T17:15:05Z","abstract_excerpt":"Large Language Models (LLMs) from the GPT family have become extremely popular, leading to a race towards reducing their inference costs to allow for efficient local computation. Yet, the vast majority of existing work focuses on weight-only quantization, which can reduce runtime costs in the memory-bound one-token-at-a-time generative setting, but does not address them in compute-bound scenarios, such as batched inference or prompt processing. In this paper, we address the general quantization problem, where both weights and activations should be quantized. We show, for the first time, that t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.09259","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.09259/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.09259","created_at":"2026-07-05T07:08:13.119055+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.09259v2","created_at":"2026-07-05T07:08:13.119055+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.09259","created_at":"2026-07-05T07:08:13.119055+00:00"},{"alias_kind":"pith_short_12","alias_value":"23BIOGC3JHG6","created_at":"2026-07-05T07:08:13.119055+00:00"},{"alias_kind":"pith_short_16","alias_value":"23BIOGC3JHG6G3C4","created_at":"2026-07-05T07:08:13.119055+00:00"},{"alias_kind":"pith_short_8","alias_value":"23BIOGC3","created_at":"2026-07-05T07:08:13.119055+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26587","citing_title":"SharQ: Bridging Activation Sparsity and FP4 Quantization for LLM Inference","ref_index":103,"is_internal_anchor":false},{"citing_arxiv_id":"2606.23419","citing_title":"GRINQH: Graded Input-based Quantization Hierarchy for Efficient LLM Generation","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/23BIOGC3JHG6G3C4PIYQSVKVZB","json":"https://pith.science/pith/23BIOGC3JHG6G3C4PIYQSVKVZB.json","graph_json":"https://pith.science/api/pith-number/23BIOGC3JHG6G3C4PIYQSVKVZB/graph.json","events_json":"https://pith.science/api/pith-number/23BIOGC3JHG6G3C4PIYQSVKVZB/events.json","paper":"https://pith.science/paper/23BIOGC3"},"agent_actions":{"view_html":"https://pith.science/pith/23BIOGC3JHG6G3C4PIYQSVKVZB","download_json":"https://pith.science/pith/23BIOGC3JHG6G3C4PIYQSVKVZB.json","view_paper":"https://pith.science/paper/23BIOGC3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.09259&json=true","fetch_graph":"https://pith.science/api/pith-number/23BIOGC3JHG6G3C4PIYQSVKVZB/graph.json","fetch_events":"https://pith.science/api/pith-number/23BIOGC3JHG6G3C4PIYQSVKVZB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/23BIOGC3JHG6G3C4PIYQSVKVZB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/23BIOGC3JHG6G3C4PIYQSVKVZB/action/storage_attestation","attest_author":"https://pith.science/pith/23BIOGC3JHG6G3C4PIYQSVKVZB/action/author_attestation","sign_citation":"https://pith.science/pith/23BIOGC3JHG6G3C4PIYQSVKVZB/action/citation_signature","submit_replication":"https://pith.science/pith/23BIOGC3JHG6G3C4PIYQSVKVZB/action/replication_record"}},"created_at":"2026-07-05T07:08:13.119055+00:00","updated_at":"2026-07-05T07:08:13.119055+00:00"}