{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:T2RWRHLVFUFC5NLKXBV45C4LBY","short_pith_number":"pith:T2RWRHLV","schema_version":"1.0","canonical_sha256":"9ea3689d752d0a2eb56ab86bce8b8b0e33e64b887ea32913c835af72cb6f6a23","source":{"kind":"arxiv","id":"2501.02423","version":3},"attestation_state":"computed","paper":{"title":"Scaling Laws for Floating Point Quantization Training","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AR","cs.CL"],"primary_cat":"cs.LG","authors_text":"An Wang, Chengzhong Xu, Di Wang, Jie Jiang, Jinbao Xue, Kan Wu, Ruobing Xie, Shuai Li, Shuaipeng Li, Weidong Han, Xingwu Sun, Yangyu Tao, Yixing Li, Yu Cheng, Zhanhui Kang, Zhen Yang","submitted_at":"2025-01-05T02:30:41Z","abstract_excerpt":"Low-precision training is considered an effective strategy for reducing both training and downstream inference costs. Previous scaling laws for precision mainly focus on integer quantization, which pay less attention to the constituents in floating-point (FP) quantization, and thus cannot well fit the LLM losses in this scenario. In contrast, while FP quantization training is more commonly implemented in production, it's research has been relatively superficial. In this paper, we thoroughly explore the effects of FP quantization targets, exponent bits, mantissa bits, and the calculation granul"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.02423","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-01-05T02:30:41Z","cross_cats_sorted":["cs.AR","cs.CL"],"title_canon_sha256":"16748ae03f1dc767f782ac5a4aae96b9c45577d49f028b80bbe7c5a08f005072","abstract_canon_sha256":"f1b1684eef08eacb1a2e2af9c4cd4d08927fe1eff7c5d9c2deb39e482a91f2de"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:15:41.578957Z","signature_b64":"gjAtZzs4OCSmDNC38sYghGjS6QiWGCnAF304GeiUBSK1SncwZQWnrVU5fVI4umBhNVKywM2lKu6T4YXClOWOCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9ea3689d752d0a2eb56ab86bce8b8b0e33e64b887ea32913c835af72cb6f6a23","last_reissued_at":"2026-07-05T11:15:41.578470Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:15:41.578470Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Scaling Laws for Floating Point Quantization Training","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AR","cs.CL"],"primary_cat":"cs.LG","authors_text":"An Wang, Chengzhong Xu, Di Wang, Jie Jiang, Jinbao Xue, Kan Wu, Ruobing Xie, Shuai Li, Shuaipeng Li, Weidong Han, Xingwu Sun, Yangyu Tao, Yixing Li, Yu Cheng, Zhanhui Kang, Zhen Yang","submitted_at":"2025-01-05T02:30:41Z","abstract_excerpt":"Low-precision training is considered an effective strategy for reducing both training and downstream inference costs. Previous scaling laws for precision mainly focus on integer quantization, which pay less attention to the constituents in floating-point (FP) quantization, and thus cannot well fit the LLM losses in this scenario. In contrast, while FP quantization training is more commonly implemented in production, it's research has been relatively superficial. In this paper, we thoroughly explore the effects of FP quantization targets, exponent bits, mantissa bits, and the calculation granul"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.02423","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.02423/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.02423","created_at":"2026-07-05T11:15:41.578530+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.02423v3","created_at":"2026-07-05T11:15:41.578530+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.02423","created_at":"2026-07-05T11:15:41.578530+00:00"},{"alias_kind":"pith_short_12","alias_value":"T2RWRHLVFUFC","created_at":"2026-07-05T11:15:41.578530+00:00"},{"alias_kind":"pith_short_16","alias_value":"T2RWRHLVFUFC5NLK","created_at":"2026-07-05T11:15:41.578530+00:00"},{"alias_kind":"pith_short_8","alias_value":"T2RWRHLV","created_at":"2026-07-05T11:15:41.578530+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/T2RWRHLVFUFC5NLKXBV45C4LBY","json":"https://pith.science/pith/T2RWRHLVFUFC5NLKXBV45C4LBY.json","graph_json":"https://pith.science/api/pith-number/T2RWRHLVFUFC5NLKXBV45C4LBY/graph.json","events_json":"https://pith.science/api/pith-number/T2RWRHLVFUFC5NLKXBV45C4LBY/events.json","paper":"https://pith.science/paper/T2RWRHLV"},"agent_actions":{"view_html":"https://pith.science/pith/T2RWRHLVFUFC5NLKXBV45C4LBY","download_json":"https://pith.science/pith/T2RWRHLVFUFC5NLKXBV45C4LBY.json","view_paper":"https://pith.science/paper/T2RWRHLV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.02423&json=true","fetch_graph":"https://pith.science/api/pith-number/T2RWRHLVFUFC5NLKXBV45C4LBY/graph.json","fetch_events":"https://pith.science/api/pith-number/T2RWRHLVFUFC5NLKXBV45C4LBY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/T2RWRHLVFUFC5NLKXBV45C4LBY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/T2RWRHLVFUFC5NLKXBV45C4LBY/action/storage_attestation","attest_author":"https://pith.science/pith/T2RWRHLVFUFC5NLKXBV45C4LBY/action/author_attestation","sign_citation":"https://pith.science/pith/T2RWRHLVFUFC5NLKXBV45C4LBY/action/citation_signature","submit_replication":"https://pith.science/pith/T2RWRHLVFUFC5NLKXBV45C4LBY/action/replication_record"}},"created_at":"2026-07-05T11:15:41.578530+00:00","updated_at":"2026-07-05T11:15:41.578530+00:00"}