{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:UE2SBYHPIXXDOBUZXW7QFK4YLR","short_pith_number":"pith:UE2SBYHP","schema_version":"1.0","canonical_sha256":"a13520e0ef45ee370699bdbf02ab985c666779618a48625c99242a90609fb1db","source":{"kind":"arxiv","id":"2411.04330","version":2},"attestation_state":"computed","paper":{"title":"Scaling Laws for Precision","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Aditi Raghunathan, Benjamin F. Spector, Blake Bordelon, Cengiz Pehlevan, Christopher R\\'e, Mansheej Paul, Niklas Muennighoff, Tanishq Kumar, Zachary Ankner","submitted_at":"2024-11-07T00:10:10Z","abstract_excerpt":"Low precision training and inference affect both the quality and cost of language models, but current scaling laws do not account for this. In this work, we devise \"precision-aware\" scaling laws for both training and inference. We propose that training in lower precision reduces the model's \"effective parameter count,\" allowing us to predict the additional loss incurred from training in low precision and post-train quantization. For inference, we find that the degradation introduced by post-training quantization increases as models are trained on more data, eventually making additional pretrai"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.04330","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-11-07T00:10:10Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"9e5ebe38a1c574b5ae9f91c5e138ef297852af018d59eaa63aeb5144993c4567","abstract_canon_sha256":"cdc6b409fbc87e04ef97235ef0b6c3732f09bfe085c09bae2b7107e4c624503d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:42:32.043489Z","signature_b64":"8XnjyUJc41pHezhKQf0uABreoAro3cvlTEOKbdffRh/ULwETwUnN/jRPyrVD7gT9FANIU6m8F2RJ3HF0Mrp5Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a13520e0ef45ee370699bdbf02ab985c666779618a48625c99242a90609fb1db","last_reissued_at":"2026-07-05T09:42:32.043010Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:42:32.043010Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Scaling Laws for Precision","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Aditi Raghunathan, Benjamin F. Spector, Blake Bordelon, Cengiz Pehlevan, Christopher R\\'e, Mansheej Paul, Niklas Muennighoff, Tanishq Kumar, Zachary Ankner","submitted_at":"2024-11-07T00:10:10Z","abstract_excerpt":"Low precision training and inference affect both the quality and cost of language models, but current scaling laws do not account for this. In this work, we devise \"precision-aware\" scaling laws for both training and inference. We propose that training in lower precision reduces the model's \"effective parameter count,\" allowing us to predict the additional loss incurred from training in low precision and post-train quantization. For inference, we find that the degradation introduced by post-training quantization increases as models are trained on more data, eventually making additional pretrai"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.04330","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.04330/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.04330","created_at":"2026-07-05T09:42:32.043086+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.04330v2","created_at":"2026-07-05T09:42:32.043086+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.04330","created_at":"2026-07-05T09:42:32.043086+00:00"},{"alias_kind":"pith_short_12","alias_value":"UE2SBYHPIXXD","created_at":"2026-07-05T09:42:32.043086+00:00"},{"alias_kind":"pith_short_16","alias_value":"UE2SBYHPIXXDOBUZ","created_at":"2026-07-05T09:42:32.043086+00:00"},{"alias_kind":"pith_short_8","alias_value":"UE2SBYHP","created_at":"2026-07-05T09:42:32.043086+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.05017","citing_title":"GoldenFloat: A Phi-Derived Static-Split Floating-Point Family from GF4 to GF1024 with a Lucas-Exact Integer Identity","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14929","citing_title":"A Hardware-Aware, Per-Layer Methodology for Post-Training Quantization of Large Language Models","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27435","citing_title":"When NPUs Are Not Always Faster: A Stage-Level Analysis of Mobile LLM Inference","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23901","citing_title":"LLMs as Noisy Channels: A Shannon Perspective on Model Capacity and Scaling Laws","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23591","citing_title":"Asymmetric Scaling Laws from Sparse Features","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2412.14590","citing_title":"MixLLM: LLM Quantization with Global Mixed-precision between Output-features and Highly-efficient System Design","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2510.18245","citing_title":"Scaling Laws Meet Model Architecture: Toward Inference-Efficient LLMs","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2603.28507","citing_title":"Continued AI Scaling Requires Repeated Efficiency Doublings","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14629","citing_title":"Switch-KD: Visual-Switch Knowledge Distillation for Vision-Language Models","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20079","citing_title":"On the Quantization Robustness of Diffusion Language Models in Coding Benchmarks","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UE2SBYHPIXXDOBUZXW7QFK4YLR","json":"https://pith.science/pith/UE2SBYHPIXXDOBUZXW7QFK4YLR.json","graph_json":"https://pith.science/api/pith-number/UE2SBYHPIXXDOBUZXW7QFK4YLR/graph.json","events_json":"https://pith.science/api/pith-number/UE2SBYHPIXXDOBUZXW7QFK4YLR/events.json","paper":"https://pith.science/paper/UE2SBYHP"},"agent_actions":{"view_html":"https://pith.science/pith/UE2SBYHPIXXDOBUZXW7QFK4YLR","download_json":"https://pith.science/pith/UE2SBYHPIXXDOBUZXW7QFK4YLR.json","view_paper":"https://pith.science/paper/UE2SBYHP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.04330&json=true","fetch_graph":"https://pith.science/api/pith-number/UE2SBYHPIXXDOBUZXW7QFK4YLR/graph.json","fetch_events":"https://pith.science/api/pith-number/UE2SBYHPIXXDOBUZXW7QFK4YLR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UE2SBYHPIXXDOBUZXW7QFK4YLR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UE2SBYHPIXXDOBUZXW7QFK4YLR/action/storage_attestation","attest_author":"https://pith.science/pith/UE2SBYHPIXXDOBUZXW7QFK4YLR/action/author_attestation","sign_citation":"https://pith.science/pith/UE2SBYHPIXXDOBUZXW7QFK4YLR/action/citation_signature","submit_replication":"https://pith.science/pith/UE2SBYHPIXXDOBUZXW7QFK4YLR/action/replication_record"}},"created_at":"2026-07-05T09:42:32.043086+00:00","updated_at":"2026-07-05T09:42:32.043086+00:00"}