{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:DQB64XYZF5M6BTDLR5UFJ6E3CV","short_pith_number":"pith:DQB64XYZ","schema_version":"1.0","canonical_sha256":"1c03ee5f192f59e0cc6b8f6854f89b157134461063c90f7c1c3e4ca9c8a8f443","source":{"kind":"arxiv","id":"2505.19115","version":2},"attestation_state":"computed","paper":{"title":"FP4 All the Way: Fully Quantized Training of LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Brian Chmiel, Daniel Soudry, Maxim Fishman, Ron Banner","submitted_at":"2025-05-25T12:14:25Z","abstract_excerpt":"We demonstrate, for the first time, fully quantized training (FQT) of large language models (LLMs) using predominantly 4-bit floating-point (FP4) precision for weights, activations, and gradients on datasets up to 200 billion tokens. We extensively investigate key design choices for FP4, including block sizes, scaling formats, and rounding methods. Our analysis shows that the NVFP4 format, where each block of 16 FP4 values (E2M1) shares a scale represented in E4M3, provides optimal results. We use stochastic rounding for backward and update passes and round-to-nearest for the forward pass to e"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.19115","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-25T12:14:25Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"74abfb16594427dd6293c8ea3dc5018428dfac57f787cec3d4e6c85b86662e96","abstract_canon_sha256":"11eed4db1c46092e6ad1466044b6f2717940bd619848aec30fe953c6ea3461c4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:51:27.346980Z","signature_b64":"+6vPSa7zawXBYyx629OxDKPUjXu2u/GRQKObN3h+ntQQXLbIVhaVpxM6naMHG33aHq99s4qS2LmInb3/vjHpAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1c03ee5f192f59e0cc6b8f6854f89b157134461063c90f7c1c3e4ca9c8a8f443","last_reissued_at":"2026-07-05T11:51:27.346189Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:51:27.346189Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FP4 All the Way: Fully Quantized Training of LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Brian Chmiel, Daniel Soudry, Maxim Fishman, Ron Banner","submitted_at":"2025-05-25T12:14:25Z","abstract_excerpt":"We demonstrate, for the first time, fully quantized training (FQT) of large language models (LLMs) using predominantly 4-bit floating-point (FP4) precision for weights, activations, and gradients on datasets up to 200 billion tokens. We extensively investigate key design choices for FP4, including block sizes, scaling formats, and rounding methods. Our analysis shows that the NVFP4 format, where each block of 16 FP4 values (E2M1) shares a scale represented in E4M3, provides optimal results. We use stochastic rounding for backward and update passes and round-to-nearest for the forward pass to e"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.19115","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.19115/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.19115","created_at":"2026-07-05T11:51:27.346286+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.19115v2","created_at":"2026-07-05T11:51:27.346286+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.19115","created_at":"2026-07-05T11:51:27.346286+00:00"},{"alias_kind":"pith_short_12","alias_value":"DQB64XYZF5M6","created_at":"2026-07-05T11:51:27.346286+00:00"},{"alias_kind":"pith_short_16","alias_value":"DQB64XYZF5M6BTDL","created_at":"2026-07-05T11:51:27.346286+00:00"},{"alias_kind":"pith_short_8","alias_value":"DQB64XYZ","created_at":"2026-07-05T11:51:27.346286+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":20,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26587","citing_title":"SharQ: Bridging Activation Sparsity and FP4 Quantization for LLM Inference","ref_index":175,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20402","citing_title":"Decomposing MXFP4 quantization error for LLM reinforcement learning: reducible bias, recoverable deadzone, and an irreducible floor","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27616","citing_title":"Not All NVFP4 QAT Recipes Are Equal: How Architecture and Scale Shape Model Quality for Anomaly Segmentation","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20402","citing_title":"Decomposing MXFP4 quantization error for LLM reinforcement learning: reducible bias, recoverable deadzone, and an irreducible floor","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23081","citing_title":"ThriftAttention: Selective Mixed Precision for Long-Context FP4 Attention","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20402","citing_title":"Decomposing MXFP4 quantization error for LLM reinforcement learning: reducible bias, recoverable deadzone, and an irreducible floor","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18739","citing_title":"LongLive-2.0: An NVFP4 Parallel Infrastructure for Long Video Generation","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2512.02010","citing_title":"Four Over Six: More Accurate NVFP4 Quantization with Adaptive Block Scaling","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2512.12131","citing_title":"BOOST: BOttleneck-Optimized Scalable Training Framework for Low-Rank Large Language Models","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2601.17187","citing_title":"High-Rate Quantized Matrix Multiplication I","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09825","citing_title":"Pretraining large language models with MXFP4 on Native FP4 Hardware","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10886","citing_title":"LoKA: Low-precision Kernel Applications for Recommendation Models At Scale","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09825","citing_title":"Pretraining large language models with MXFP4 on Native FP4 Hardware","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02525","citing_title":"AdaHOP: Fast and Accurate Low-Precision Training via Outlier-Pattern-Aware Rotation","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12327","citing_title":"Grid Games: The Power of Multiple Grids for Quantizing Large Language Models","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12464","citing_title":"Search Your Block Floating Point Scales!","ref_index":109,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10886","citing_title":"LoKA: Low-precision Kernel Applications for Recommendation Models At Scale","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09825","citing_title":"Pretraining large language models with MXFP4 on Native FP4 Hardware","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08826","citing_title":"HiFloat4 Format for Language Model Pre-training on Ascend NPUs","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04523","citing_title":"LOCALUT: Harnessing Capacity-Computation Tradeoffs for LUT-Based Inference in DRAM-PIM","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DQB64XYZF5M6BTDLR5UFJ6E3CV","json":"https://pith.science/pith/DQB64XYZF5M6BTDLR5UFJ6E3CV.json","graph_json":"https://pith.science/api/pith-number/DQB64XYZF5M6BTDLR5UFJ6E3CV/graph.json","events_json":"https://pith.science/api/pith-number/DQB64XYZF5M6BTDLR5UFJ6E3CV/events.json","paper":"https://pith.science/paper/DQB64XYZ"},"agent_actions":{"view_html":"https://pith.science/pith/DQB64XYZF5M6BTDLR5UFJ6E3CV","download_json":"https://pith.science/pith/DQB64XYZF5M6BTDLR5UFJ6E3CV.json","view_paper":"https://pith.science/paper/DQB64XYZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.19115&json=true","fetch_graph":"https://pith.science/api/pith-number/DQB64XYZF5M6BTDLR5UFJ6E3CV/graph.json","fetch_events":"https://pith.science/api/pith-number/DQB64XYZF5M6BTDLR5UFJ6E3CV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DQB64XYZF5M6BTDLR5UFJ6E3CV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DQB64XYZF5M6BTDLR5UFJ6E3CV/action/storage_attestation","attest_author":"https://pith.science/pith/DQB64XYZF5M6BTDLR5UFJ6E3CV/action/author_attestation","sign_citation":"https://pith.science/pith/DQB64XYZF5M6BTDLR5UFJ6E3CV/action/citation_signature","submit_replication":"https://pith.science/pith/DQB64XYZF5M6BTDLR5UFJ6E3CV/action/replication_record"}},"created_at":"2026-07-05T11:51:27.346286+00:00","updated_at":"2026-07-05T11:51:27.346286+00:00"}