{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:FI2QGB4SC2WGXE6PSWLUC7SD2K","short_pith_number":"pith:FI2QGB4S","schema_version":"1.0","canonical_sha256":"2a3503079216ac6b93cf9597417e43d2b533cf340b4a93a0dd81c353e22e4154","source":{"kind":"arxiv","id":"2502.20566","version":1},"attestation_state":"computed","paper":{"title":"Stochastic Rounding for LLM Training: Theory and Practice","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Kaan Ozkara, Tao Yu, Youngsuk Park","submitted_at":"2025-02-27T22:08:08Z","abstract_excerpt":"As the parameters of Large Language Models (LLMs) have scaled to hundreds of billions, the demand for efficient training methods -- balancing faster computation and reduced memory usage without sacrificing accuracy -- has become more critical than ever. In recent years, various mixed precision strategies, which involve different precision levels for optimization components, have been proposed to increase training speed with minimal accuracy degradation. However, these strategies often require manual adjustments and lack theoretical justification. In this work, we leverage stochastic rounding ("},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.20566","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-27T22:08:08Z","cross_cats_sorted":[],"title_canon_sha256":"73d70a2097a4081a56c461d26c7f3eefde94ba1270ecdf24a7949e2aeb23cea2","abstract_canon_sha256":"3b341e4d813fdc07785035adeec8ccabf9d81fd7606977bb4589de9c1aa11e56"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:21:35.423795Z","signature_b64":"DtT1Xxc8aMkrc0q+knUaFgEBeBGETcFJn+CK8SqSpkkXKRLLxe73XGe3rndzHaCV4jOZE0eQipLMdPKOvmeqBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2a3503079216ac6b93cf9597417e43d2b533cf340b4a93a0dd81c353e22e4154","last_reissued_at":"2026-07-05T10:21:35.422908Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:21:35.422908Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Stochastic Rounding for LLM Training: Theory and Practice","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Kaan Ozkara, Tao Yu, Youngsuk Park","submitted_at":"2025-02-27T22:08:08Z","abstract_excerpt":"As the parameters of Large Language Models (LLMs) have scaled to hundreds of billions, the demand for efficient training methods -- balancing faster computation and reduced memory usage without sacrificing accuracy -- has become more critical than ever. In recent years, various mixed precision strategies, which involve different precision levels for optimization components, have been proposed to increase training speed with minimal accuracy degradation. However, these strategies often require manual adjustments and lack theoretical justification. In this work, we leverage stochastic rounding ("},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.20566","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.20566/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.20566","created_at":"2026-07-05T10:21:35.422984+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.20566v1","created_at":"2026-07-05T10:21:35.422984+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.20566","created_at":"2026-07-05T10:21:35.422984+00:00"},{"alias_kind":"pith_short_12","alias_value":"FI2QGB4SC2WG","created_at":"2026-07-05T10:21:35.422984+00:00"},{"alias_kind":"pith_short_16","alias_value":"FI2QGB4SC2WGXE6P","created_at":"2026-07-05T10:21:35.422984+00:00"},{"alias_kind":"pith_short_8","alias_value":"FI2QGB4S","created_at":"2026-07-05T10:21:35.422984+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.17064","citing_title":"Towards Human-Level Book-Writing Capability","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17064","citing_title":"Towards Human-Level Book-Writing Capability","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2409.02977","citing_title":"Large Language Model-Based Agents for Software Engineering: A Survey","ref_index":257,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FI2QGB4SC2WGXE6PSWLUC7SD2K","json":"https://pith.science/pith/FI2QGB4SC2WGXE6PSWLUC7SD2K.json","graph_json":"https://pith.science/api/pith-number/FI2QGB4SC2WGXE6PSWLUC7SD2K/graph.json","events_json":"https://pith.science/api/pith-number/FI2QGB4SC2WGXE6PSWLUC7SD2K/events.json","paper":"https://pith.science/paper/FI2QGB4S"},"agent_actions":{"view_html":"https://pith.science/pith/FI2QGB4SC2WGXE6PSWLUC7SD2K","download_json":"https://pith.science/pith/FI2QGB4SC2WGXE6PSWLUC7SD2K.json","view_paper":"https://pith.science/paper/FI2QGB4S","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.20566&json=true","fetch_graph":"https://pith.science/api/pith-number/FI2QGB4SC2WGXE6PSWLUC7SD2K/graph.json","fetch_events":"https://pith.science/api/pith-number/FI2QGB4SC2WGXE6PSWLUC7SD2K/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FI2QGB4SC2WGXE6PSWLUC7SD2K/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FI2QGB4SC2WGXE6PSWLUC7SD2K/action/storage_attestation","attest_author":"https://pith.science/pith/FI2QGB4SC2WGXE6PSWLUC7SD2K/action/author_attestation","sign_citation":"https://pith.science/pith/FI2QGB4SC2WGXE6PSWLUC7SD2K/action/citation_signature","submit_replication":"https://pith.science/pith/FI2QGB4SC2WGXE6PSWLUC7SD2K/action/replication_record"}},"created_at":"2026-07-05T10:21:35.422984+00:00","updated_at":"2026-07-05T10:21:35.422984+00:00"}