{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:SGZ57SD7DMJTLISNYY2HWNLQPH","short_pith_number":"pith:SGZ57SD7","schema_version":"1.0","canonical_sha256":"91b3dfc87f1b1335a24dc6347b357079f77f54ca9b98122bf15209e32267bbaa","source":{"kind":"arxiv","id":"2502.20586","version":3},"attestation_state":"computed","paper":{"title":"Training LLMs with MXFP4","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Albert Tseng, Tao Yu, Youngsuk Park","submitted_at":"2025-02-27T23:01:31Z","abstract_excerpt":"Low precision (LP) datatypes such as MXFP4 can accelerate matrix multiplications (GEMMs) and reduce training costs. However, directly using MXFP4 instead of BF16 during training significantly degrades model quality. In this work, we present the first near-lossless training recipe that uses MXFP4 GEMMs, which are $2\\times$ faster than FP8 on supported hardware. Our key insight is to compute unbiased gradient estimates with stochastic rounding (SR), resulting in more accurate model updates. However, directly applying SR to MXFP4 can result in high variance from block-level outliers, harming conv"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.20586","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-27T23:01:31Z","cross_cats_sorted":[],"title_canon_sha256":"a0f5d050bb1fe9b5a2e2d8bf318427798fa48e8d52be52190c0fbfb4d109af70","abstract_canon_sha256":"37d11004505631b77cc250a6199ce6703c1a80c9b891ff52097619bdf6e43a78"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:59:57.960185Z","signature_b64":"MKE/Z8iqjBmEpNGWHmYRUkKB736yj7xk0D28h/4yAUldBaXNy69Ro0OLkW/iXtSwa3gWu1Hd29bUqhw67Q75Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"91b3dfc87f1b1335a24dc6347b357079f77f54ca9b98122bf15209e32267bbaa","last_reissued_at":"2026-07-05T11:59:57.959721Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:59:57.959721Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Training LLMs with MXFP4","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Albert Tseng, Tao Yu, Youngsuk Park","submitted_at":"2025-02-27T23:01:31Z","abstract_excerpt":"Low precision (LP) datatypes such as MXFP4 can accelerate matrix multiplications (GEMMs) and reduce training costs. However, directly using MXFP4 instead of BF16 during training significantly degrades model quality. In this work, we present the first near-lossless training recipe that uses MXFP4 GEMMs, which are $2\\times$ faster than FP8 on supported hardware. Our key insight is to compute unbiased gradient estimates with stochastic rounding (SR), resulting in more accurate model updates. However, directly applying SR to MXFP4 can result in high variance from block-level outliers, harming conv"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.20586","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.20586/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.20586","created_at":"2026-07-05T11:59:57.959780+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.20586v3","created_at":"2026-07-05T11:59:57.959780+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.20586","created_at":"2026-07-05T11:59:57.959780+00:00"},{"alias_kind":"pith_short_12","alias_value":"SGZ57SD7DMJT","created_at":"2026-07-05T11:59:57.959780+00:00"},{"alias_kind":"pith_short_16","alias_value":"SGZ57SD7DMJTLISN","created_at":"2026-07-05T11:59:57.959780+00:00"},{"alias_kind":"pith_short_8","alias_value":"SGZ57SD7","created_at":"2026-07-05T11:59:57.959780+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.00703","citing_title":"Information-Theoretic Lower Bounds for Bit-Constrained Stochastic Optimization via a Reduction to Compressed Gaussian Mean Estimation","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00312","citing_title":"Stochastic Rounding Increases Small Singular Values","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20402","citing_title":"Decomposing MXFP4 quantization error for LLM reinforcement learning: reducible bias, recoverable deadzone, and an irreducible floor","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20402","citing_title":"Decomposing MXFP4 quantization error for LLM reinforcement learning: reducible bias, recoverable deadzone, and an irreducible floor","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20402","citing_title":"Decomposing MXFP4 quantization error for LLM reinforcement learning: reducible bias, recoverable deadzone, and an irreducible floor","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2510.04212","citing_title":"Why Low-Precision Transformer Training Fails: An Analysis on Flash Attention","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2512.02010","citing_title":"Four Over Six: More Accurate NVFP4 Quantization with Adaptive Block Scaling","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12327","citing_title":"Grid Games: The Power of Multiple Grids for Quantizing Large Language Models","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12464","citing_title":"Search Your Block Floating Point Scales!","ref_index":158,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12798","citing_title":"VFA: Relieving Vector Operations in Flash Attention with Global Maximum Pre-computation","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SGZ57SD7DMJTLISNYY2HWNLQPH","json":"https://pith.science/pith/SGZ57SD7DMJTLISNYY2HWNLQPH.json","graph_json":"https://pith.science/api/pith-number/SGZ57SD7DMJTLISNYY2HWNLQPH/graph.json","events_json":"https://pith.science/api/pith-number/SGZ57SD7DMJTLISNYY2HWNLQPH/events.json","paper":"https://pith.science/paper/SGZ57SD7"},"agent_actions":{"view_html":"https://pith.science/pith/SGZ57SD7DMJTLISNYY2HWNLQPH","download_json":"https://pith.science/pith/SGZ57SD7DMJTLISNYY2HWNLQPH.json","view_paper":"https://pith.science/paper/SGZ57SD7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.20586&json=true","fetch_graph":"https://pith.science/api/pith-number/SGZ57SD7DMJTLISNYY2HWNLQPH/graph.json","fetch_events":"https://pith.science/api/pith-number/SGZ57SD7DMJTLISNYY2HWNLQPH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SGZ57SD7DMJTLISNYY2HWNLQPH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SGZ57SD7DMJTLISNYY2HWNLQPH/action/storage_attestation","attest_author":"https://pith.science/pith/SGZ57SD7DMJTLISNYY2HWNLQPH/action/author_attestation","sign_citation":"https://pith.science/pith/SGZ57SD7DMJTLISNYY2HWNLQPH/action/citation_signature","submit_replication":"https://pith.science/pith/SGZ57SD7DMJTLISNYY2HWNLQPH/action/replication_record"}},"created_at":"2026-07-05T11:59:57.959780+00:00","updated_at":"2026-07-05T11:59:57.959780+00:00"}