{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:QEL4HKLMTI4QJMQL32TN7UQGNO","short_pith_number":"pith:QEL4HKLM","schema_version":"1.0","canonical_sha256":"8117c3a96c9a3904b20bdea6dfd2066b9d7f70c7509f056aae77454796cbd03c","source":{"kind":"arxiv","id":"2306.07629","version":4},"attestation_state":"computed","paper":{"title":"SqueezeLLM: Dense-and-Sparse Quantization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Amir Gholami, Coleman Hooper, Kurt Keutzer, Michael W. Mahoney, Sehoon Kim, Sheng Shen, Xiuyu Li, Zhen Dong","submitted_at":"2023-06-13T08:57:54Z","abstract_excerpt":"Generative Large Language Models (LLMs) have demonstrated remarkable results for a wide range of tasks. However, deploying these models for inference has been a significant challenge due to their unprecedented resource requirements. This has forced existing deployment frameworks to use multi-GPU inference pipelines, which are often complex and costly, or to use smaller and less performant models. In this work, we demonstrate that the main bottleneck for generative inference with LLMs is memory bandwidth, rather than compute, specifically for single batch inference. While quantization has emerg"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.07629","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-06-13T08:57:54Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"acec35eb64829b1a36c30afb3e7f970cec7326eefafea01465e7ae29805a157f","abstract_canon_sha256":"ff25315d6c87b11a6d20b92d2ebc27bb82fe942a3763acf57869dd844c81cdfa"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:27:22.509187Z","signature_b64":"1DCIEXPOHds+nXnGyk6qpv+RdYjrPDvZndgStUZlrTVaIZgc+3oYJzZ28PGFexEzRd8S1YTMKMhIHFaWRc2ACQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8117c3a96c9a3904b20bdea6dfd2066b9d7f70c7509f056aae77454796cbd03c","last_reissued_at":"2026-07-05T08:27:22.508774Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:27:22.508774Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SqueezeLLM: Dense-and-Sparse Quantization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Amir Gholami, Coleman Hooper, Kurt Keutzer, Michael W. Mahoney, Sehoon Kim, Sheng Shen, Xiuyu Li, Zhen Dong","submitted_at":"2023-06-13T08:57:54Z","abstract_excerpt":"Generative Large Language Models (LLMs) have demonstrated remarkable results for a wide range of tasks. However, deploying these models for inference has been a significant challenge due to their unprecedented resource requirements. This has forced existing deployment frameworks to use multi-GPU inference pipelines, which are often complex and costly, or to use smaller and less performant models. In this work, we demonstrate that the main bottleneck for generative inference with LLMs is memory bandwidth, rather than compute, specifically for single batch inference. While quantization has emerg"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.07629","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.07629/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.07629","created_at":"2026-07-05T08:27:22.508831+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.07629v4","created_at":"2026-07-05T08:27:22.508831+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.07629","created_at":"2026-07-05T08:27:22.508831+00:00"},{"alias_kind":"pith_short_12","alias_value":"QEL4HKLMTI4Q","created_at":"2026-07-05T08:27:22.508831+00:00"},{"alias_kind":"pith_short_16","alias_value":"QEL4HKLMTI4QJMQL","created_at":"2026-07-05T08:27:22.508831+00:00"},{"alias_kind":"pith_short_8","alias_value":"QEL4HKLM","created_at":"2026-07-05T08:27:22.508831+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":31,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01876","citing_title":"SAB-LVLM: Significance-Aware Binarization for Large Vision-Language Models","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2607.02461","citing_title":"OrbitQuant: Data-Agnostic Quantization for Image and Video Diffusion Transformers","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01065","citing_title":"GSRQ: Gain-Shape Residual Quantization for Sub-1-bit KV Cache","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05429","citing_title":"Minimizing the Hidden Cost of Scales: Graph-Guided Ultra-Low-Bit Quantization for Large Language Models","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01412","citing_title":"GPTQ-intrinsic LoRA: A Near-optimal Algorithm for Low-precision Quantization with Low-rank Adaptation","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28438","citing_title":"When AI Reviews Its Own Code: Recursive Self-Training Collapse in Code LLMs","ref_index":141,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14844","citing_title":"XFP: Quality-Targeted Adaptive Codebook Quantization with Sparse Outlier Separation for LLM Inference","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24011","citing_title":"ActQuant: Sub-4-bit Action-Guided Quantization for Vision-Language-Action Models","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00535","citing_title":"DREAM-S: Speculative Decoding with Searchable Drafting and Target-Aware Refinement for Multimodal Generation","ref_index":73,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23078","citing_title":"GEMQ: Global Expert-Level Mixed-Precision Quantization for MoE LLMs","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2503.08223","citing_title":"Will LLMs Scaling Hit the Wall? Breaking Barriers via Distributed Resources on Massive Edge Devices","ref_index":174,"is_internal_anchor":false},{"citing_arxiv_id":"2505.02380","citing_title":"EntroLLM: Entropy Encoded Weight Compression for Efficient Large Language Model Inference on Edge Devices","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2312.05821","citing_title":"ASVD: Activation-aware Singular Value Decomposition for Compressing Large Language Models","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2504.19874","citing_title":"TurboQuant: Online Vector Quantization with Near-optimal Distortion Rate","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19929","citing_title":"Breaking Modality Heterogeneity in Low-Bit Quantization for Large Vision-Language Models","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2506.13727","citing_title":"Attribution-Guided Pruning for Insight and Control: Circuit Discovery and Targeted Correction in Small-scale LLMs","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2403.12031","citing_title":"RouterBench: A Benchmark for Multi-LLM Routing System","ref_index":89,"is_internal_anchor":false},{"citing_arxiv_id":"2602.05902","citing_title":"CoreQ: Learning-Free Mismatch Correction and Successive Rounding for Quantization","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2405.16406","citing_title":"SpinQuant: LLM quantization with learned rotations","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2404.14294","citing_title":"A Survey on Efficient Inference for Large Language Models","ref_index":197,"is_internal_anchor":false},{"citing_arxiv_id":"2401.10774","citing_title":"Medusa: Simple LLM Inference Acceleration Framework with Multiple Decoding Heads","ref_index":81,"is_internal_anchor":false},{"citing_arxiv_id":"2402.02750","citing_title":"KIVI: A Tuning-Free Asymmetric 2bit Quantization for KV Cache","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04738","citing_title":"OSAQ: Outlier Self-Absorption for Accurate Low-bit LLM Quantization","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24273","citing_title":"BitRL: Reinforcement Learning with 1-bit Quantized Language Models for Resource-Constrained Edge Deployment","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24008","citing_title":"Coverage-Based Calibration for Post-Training Quantization via Weighted Set Cover over Outlier Channels","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QEL4HKLMTI4QJMQL32TN7UQGNO","json":"https://pith.science/pith/QEL4HKLMTI4QJMQL32TN7UQGNO.json","graph_json":"https://pith.science/api/pith-number/QEL4HKLMTI4QJMQL32TN7UQGNO/graph.json","events_json":"https://pith.science/api/pith-number/QEL4HKLMTI4QJMQL32TN7UQGNO/events.json","paper":"https://pith.science/paper/QEL4HKLM"},"agent_actions":{"view_html":"https://pith.science/pith/QEL4HKLMTI4QJMQL32TN7UQGNO","download_json":"https://pith.science/pith/QEL4HKLMTI4QJMQL32TN7UQGNO.json","view_paper":"https://pith.science/paper/QEL4HKLM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.07629&json=true","fetch_graph":"https://pith.science/api/pith-number/QEL4HKLMTI4QJMQL32TN7UQGNO/graph.json","fetch_events":"https://pith.science/api/pith-number/QEL4HKLMTI4QJMQL32TN7UQGNO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QEL4HKLMTI4QJMQL32TN7UQGNO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QEL4HKLMTI4QJMQL32TN7UQGNO/action/storage_attestation","attest_author":"https://pith.science/pith/QEL4HKLMTI4QJMQL32TN7UQGNO/action/author_attestation","sign_citation":"https://pith.science/pith/QEL4HKLMTI4QJMQL32TN7UQGNO/action/citation_signature","submit_replication":"https://pith.science/pith/QEL4HKLMTI4QJMQL32TN7UQGNO/action/replication_record"}},"created_at":"2026-07-05T08:27:22.508831+00:00","updated_at":"2026-07-05T08:27:22.508831+00:00"}