{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:V42TV3EHR56E4MRQ4XHC6AOM45","short_pith_number":"pith:V42TV3EH","schema_version":"1.0","canonical_sha256":"af353aec878f7c4e3230e5ce2f01cce74ffb943048262e48d8468e78597f46dc","source":{"kind":"arxiv","id":"2505.14302","version":1},"attestation_state":"computed","paper":{"title":"Scaling Law for Quantization-Aware Training","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Chaoyi Zhang, Jie Huang, Jing Liu, Jin Ma, Mengzhao Chen, Ping Luo, Xun Zhou, Yunshui Li, Yutao Zeng, Zeyue Xue, Zhiheng Liu","submitted_at":"2025-05-20T12:54:43Z","abstract_excerpt":"Large language models (LLMs) demand substantial computational and memory resources, creating deployment challenges. Quantization-aware training (QAT) addresses these challenges by reducing model precision while maintaining performance. However, the scaling behavior of QAT, especially at 4-bit precision (W4A4), is not well understood. Existing QAT scaling laws often ignore key factors such as the number of training tokens and quantization granularity, which limits their applicability. This paper proposes a unified scaling law for QAT that models quantization error as a function of model size, t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.14302","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-20T12:54:43Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"da187b39437d4733126a2089cbb0ea8e472beb974327b833013ab673e7f08d14","abstract_canon_sha256":"a9f0062e7413492fc7717a1a973c7ee33b6be794b6ab7fb395c78535759fa6be"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:05:59.366611Z","signature_b64":"+wkm4qFW85mtKH4AmVmLNFuTvCp2zY5EWJ9+jCgnOHeMSlPXqyHbyrfZsRD36vu9NuagXysDJnDicOOa3GkRBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"af353aec878f7c4e3230e5ce2f01cce74ffb943048262e48d8468e78597f46dc","last_reissued_at":"2026-07-05T11:05:59.366144Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:05:59.366144Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Scaling Law for Quantization-Aware Training","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Chaoyi Zhang, Jie Huang, Jing Liu, Jin Ma, Mengzhao Chen, Ping Luo, Xun Zhou, Yunshui Li, Yutao Zeng, Zeyue Xue, Zhiheng Liu","submitted_at":"2025-05-20T12:54:43Z","abstract_excerpt":"Large language models (LLMs) demand substantial computational and memory resources, creating deployment challenges. Quantization-aware training (QAT) addresses these challenges by reducing model precision while maintaining performance. However, the scaling behavior of QAT, especially at 4-bit precision (W4A4), is not well understood. Existing QAT scaling laws often ignore key factors such as the number of training tokens and quantization granularity, which limits their applicability. This paper proposes a unified scaling law for QAT that models quantization error as a function of model size, t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.14302","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.14302/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.14302","created_at":"2026-07-05T11:05:59.366201+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.14302v1","created_at":"2026-07-05T11:05:59.366201+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.14302","created_at":"2026-07-05T11:05:59.366201+00:00"},{"alias_kind":"pith_short_12","alias_value":"V42TV3EHR56E","created_at":"2026-07-05T11:05:59.366201+00:00"},{"alias_kind":"pith_short_16","alias_value":"V42TV3EHR56E4MRQ","created_at":"2026-07-05T11:05:59.366201+00:00"},{"alias_kind":"pith_short_8","alias_value":"V42TV3EH","created_at":"2026-07-05T11:05:59.366201+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.08761","citing_title":"APEX4: Efficient Pure W4A4 LLM Inference via Intra-SM Compute Rebalancing","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08761","citing_title":"APEX4: Efficient Pure W4A4 LLM Inference via Intra-SM Compute Rebalancing","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2510.18245","citing_title":"Scaling Laws Meet Model Architecture: Toward Inference-Efficient LLMs","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2512.20856","citing_title":"NVIDIA Nemotron 3: Efficient and Open Intelligence","ref_index":197,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15167","citing_title":"When Flat Minima Fail: Characterizing INT4 Quantization Collapse After FP32 Convergence","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/V42TV3EHR56E4MRQ4XHC6AOM45","json":"https://pith.science/pith/V42TV3EHR56E4MRQ4XHC6AOM45.json","graph_json":"https://pith.science/api/pith-number/V42TV3EHR56E4MRQ4XHC6AOM45/graph.json","events_json":"https://pith.science/api/pith-number/V42TV3EHR56E4MRQ4XHC6AOM45/events.json","paper":"https://pith.science/paper/V42TV3EH"},"agent_actions":{"view_html":"https://pith.science/pith/V42TV3EHR56E4MRQ4XHC6AOM45","download_json":"https://pith.science/pith/V42TV3EHR56E4MRQ4XHC6AOM45.json","view_paper":"https://pith.science/paper/V42TV3EH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.14302&json=true","fetch_graph":"https://pith.science/api/pith-number/V42TV3EHR56E4MRQ4XHC6AOM45/graph.json","fetch_events":"https://pith.science/api/pith-number/V42TV3EHR56E4MRQ4XHC6AOM45/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/V42TV3EHR56E4MRQ4XHC6AOM45/action/timestamp_anchor","attest_storage":"https://pith.science/pith/V42TV3EHR56E4MRQ4XHC6AOM45/action/storage_attestation","attest_author":"https://pith.science/pith/V42TV3EHR56E4MRQ4XHC6AOM45/action/author_attestation","sign_citation":"https://pith.science/pith/V42TV3EHR56E4MRQ4XHC6AOM45/action/citation_signature","submit_replication":"https://pith.science/pith/V42TV3EHR56E4MRQ4XHC6AOM45/action/replication_record"}},"created_at":"2026-07-05T11:05:59.366201+00:00","updated_at":"2026-07-05T11:05:59.366201+00:00"}