{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:M7ZCTMLJKDTJMTQFCC4UV762PW","short_pith_number":"pith:M7ZCTMLJ","schema_version":"1.0","canonical_sha256":"67f229b16950e6964e0510b94affda7d86f7e3548b75d193d3ee89b6b5a682bc","source":{"kind":"arxiv","id":"2411.17691","version":2},"attestation_state":"computed","paper":{"title":"Low-Bit Quantization Favors Undertrained LLMs: Scaling Laws for Quantized LLMs with 100T Training Tokens","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Dong Yu, Haitao Mi, Tao Ge, Thomas Hartvigsen, Xu Ouyang, Zhisong Zhang","submitted_at":"2024-11-26T18:57:58Z","abstract_excerpt":"We reveal that low-bit quantization favors undertrained large language models (LLMs) by observing that models with larger sizes or fewer training tokens experience less quantization-induced degradation (QiD) when applying low-bit quantization, whereas smaller models with extensive training tokens suffer significant QiD. To gain deeper insights into this trend, we study over 1500 quantized LLM checkpoints of various sizes and at different training levels (undertrained or fully trained) in a controlled setting, deriving scaling laws for understanding the relationship between QiD and factors such"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.17691","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-11-26T18:57:58Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"a9be467638c8cc37a05a7e2a28ba3122865227e635c6230ab5c20d90a12072be","abstract_canon_sha256":"04951414061653964bf59cdffcfe906648069f522c80810a50f379f38b6abd26"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:41:05.137940Z","signature_b64":"mwY6eCCqRK36hyMKY5BWeJL6sO4pHg0SBU6jiC3bctq5mI1RupAF3zx3Z4Mv9f5QfIhXYqrugy5NRLb8TJXXBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"67f229b16950e6964e0510b94affda7d86f7e3548b75d193d3ee89b6b5a682bc","last_reissued_at":"2026-07-05T09:41:05.137444Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:41:05.137444Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Low-Bit Quantization Favors Undertrained LLMs: Scaling Laws for Quantized LLMs with 100T Training Tokens","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Dong Yu, Haitao Mi, Tao Ge, Thomas Hartvigsen, Xu Ouyang, Zhisong Zhang","submitted_at":"2024-11-26T18:57:58Z","abstract_excerpt":"We reveal that low-bit quantization favors undertrained large language models (LLMs) by observing that models with larger sizes or fewer training tokens experience less quantization-induced degradation (QiD) when applying low-bit quantization, whereas smaller models with extensive training tokens suffer significant QiD. To gain deeper insights into this trend, we study over 1500 quantized LLM checkpoints of various sizes and at different training levels (undertrained or fully trained) in a controlled setting, deriving scaling laws for understanding the relationship between QiD and factors such"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.17691","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.17691/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.17691","created_at":"2026-07-05T09:41:05.137503+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.17691v2","created_at":"2026-07-05T09:41:05.137503+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.17691","created_at":"2026-07-05T09:41:05.137503+00:00"},{"alias_kind":"pith_short_12","alias_value":"M7ZCTMLJKDTJ","created_at":"2026-07-05T09:41:05.137503+00:00"},{"alias_kind":"pith_short_16","alias_value":"M7ZCTMLJKDTJMTQF","created_at":"2026-07-05T09:41:05.137503+00:00"},{"alias_kind":"pith_short_8","alias_value":"M7ZCTMLJ","created_at":"2026-07-05T09:41:05.137503+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22249","citing_title":"On the Expressive Power of Weight Quantization in Large Language Models","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23901","citing_title":"LLMs as Noisy Channels: A Shannon Perspective on Model Capacity and Scaling Laws","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2508.06974","citing_title":"Rethinking 1-bit Optimization Leveraging Pre-trained Large Language Models","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21171","citing_title":"FTerViT: Fully Ternary Vision Transformer","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24273","citing_title":"BitRL: Reinforcement Learning with 1-bit Quantized Language Models for Resource-Constrained Edge Deployment","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21254","citing_title":"Hyperloop Transformers","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19167","citing_title":"LBLLM: Lightweight Binarization of Large Language Models via Three-Stage Distillation","ref_index":100,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/M7ZCTMLJKDTJMTQFCC4UV762PW","json":"https://pith.science/pith/M7ZCTMLJKDTJMTQFCC4UV762PW.json","graph_json":"https://pith.science/api/pith-number/M7ZCTMLJKDTJMTQFCC4UV762PW/graph.json","events_json":"https://pith.science/api/pith-number/M7ZCTMLJKDTJMTQFCC4UV762PW/events.json","paper":"https://pith.science/paper/M7ZCTMLJ"},"agent_actions":{"view_html":"https://pith.science/pith/M7ZCTMLJKDTJMTQFCC4UV762PW","download_json":"https://pith.science/pith/M7ZCTMLJKDTJMTQFCC4UV762PW.json","view_paper":"https://pith.science/paper/M7ZCTMLJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.17691&json=true","fetch_graph":"https://pith.science/api/pith-number/M7ZCTMLJKDTJMTQFCC4UV762PW/graph.json","fetch_events":"https://pith.science/api/pith-number/M7ZCTMLJKDTJMTQFCC4UV762PW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/M7ZCTMLJKDTJMTQFCC4UV762PW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/M7ZCTMLJKDTJMTQFCC4UV762PW/action/storage_attestation","attest_author":"https://pith.science/pith/M7ZCTMLJKDTJMTQFCC4UV762PW/action/author_attestation","sign_citation":"https://pith.science/pith/M7ZCTMLJKDTJMTQFCC4UV762PW/action/citation_signature","submit_replication":"https://pith.science/pith/M7ZCTMLJKDTJMTQFCC4UV762PW/action/replication_record"}},"created_at":"2026-07-05T09:41:05.137503+00:00","updated_at":"2026-07-05T09:41:05.137503+00:00"}