{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:PNAGIFK6FBFCOXJYS7EOJKGTZP","short_pith_number":"pith:PNAGIFK6","schema_version":"1.0","canonical_sha256":"7b4064155e284a275d3897c8e4a8d3cbfb2afb9d30e5172eba60c73dc9a03388","source":{"kind":"arxiv","id":"2206.02915","version":1},"attestation_state":"computed","paper":{"title":"8-bit Numerical Formats for Deep Neural Networks","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Badreddine Noune, Carlo Luschi, Daniel Justus, Dominic Masters, Philip Jones","submitted_at":"2022-06-06T21:31:32Z","abstract_excerpt":"Given the current trend of increasing size and complexity of machine learning architectures, it has become of critical importance to identify new approaches to improve the computational efficiency of model training. In this context, we address the advantages of floating-point over fixed-point representation, and present an in-depth study on the use of 8-bit floating-point number formats for activations, weights, and gradients for both training and inference. We explore the effect of different bit-widths for exponents and significands and different exponent biases. The experimental results demo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2206.02915","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2022-06-06T21:31:32Z","cross_cats_sorted":[],"title_canon_sha256":"d22697a129ddc9600d2242ab2a4dfeb9063f10da5d96b102160e166bc3a0804b","abstract_canon_sha256":"1134156ba0cf5aaf14e816e146c117493be361b4a8e37e47b440bcd903fed274"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:29:37.116189Z","signature_b64":"dOKvGBK03IJ+arkvRSVwb8B4OERg6Jfc2PopZX3b+s/mf1xMGF36+UojV/ijgqqn85hDCUGQE3iPH4ZKvZEFBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7b4064155e284a275d3897c8e4a8d3cbfb2afb9d30e5172eba60c73dc9a03388","last_reissued_at":"2026-07-05T04:29:37.115574Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:29:37.115574Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"8-bit Numerical Formats for Deep Neural Networks","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Badreddine Noune, Carlo Luschi, Daniel Justus, Dominic Masters, Philip Jones","submitted_at":"2022-06-06T21:31:32Z","abstract_excerpt":"Given the current trend of increasing size and complexity of machine learning architectures, it has become of critical importance to identify new approaches to improve the computational efficiency of model training. In this context, we address the advantages of floating-point over fixed-point representation, and present an in-depth study on the use of 8-bit floating-point number formats for activations, weights, and gradients for both training and inference. We explore the effect of different bit-widths for exponents and significands and different exponent biases. The experimental results demo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2206.02915","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2206.02915/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2206.02915","created_at":"2026-07-05T04:29:37.115678+00:00"},{"alias_kind":"arxiv_version","alias_value":"2206.02915v1","created_at":"2026-07-05T04:29:37.115678+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2206.02915","created_at":"2026-07-05T04:29:37.115678+00:00"},{"alias_kind":"pith_short_12","alias_value":"PNAGIFK6FBFC","created_at":"2026-07-05T04:29:37.115678+00:00"},{"alias_kind":"pith_short_16","alias_value":"PNAGIFK6FBFCOXJY","created_at":"2026-07-05T04:29:37.115678+00:00"},{"alias_kind":"pith_short_8","alias_value":"PNAGIFK6","created_at":"2026-07-05T04:29:37.115678+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09686","citing_title":"An 83-Format Numeric Catalog with Bit-Exact Conformance Vectors: A Vendor-Neutral Reference for FP8, BF16, MXFP4, and Microscaling Formats","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04028","citing_title":"Novel Aspects of IEEE SA P3109 Arithmetic Formats for Machine Learning","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2510.04212","citing_title":"Why Low-Precision Transformer Training Fails: An Analysis on Flash Attention","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2209.05433","citing_title":"FP8 Formats for Deep Learning","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15416","citing_title":"StoSignSGD: Unbiased Structural Stochasticity Fixes SignSGD for Training Large Language Models","ref_index":28,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PNAGIFK6FBFCOXJYS7EOJKGTZP","json":"https://pith.science/pith/PNAGIFK6FBFCOXJYS7EOJKGTZP.json","graph_json":"https://pith.science/api/pith-number/PNAGIFK6FBFCOXJYS7EOJKGTZP/graph.json","events_json":"https://pith.science/api/pith-number/PNAGIFK6FBFCOXJYS7EOJKGTZP/events.json","paper":"https://pith.science/paper/PNAGIFK6"},"agent_actions":{"view_html":"https://pith.science/pith/PNAGIFK6FBFCOXJYS7EOJKGTZP","download_json":"https://pith.science/pith/PNAGIFK6FBFCOXJYS7EOJKGTZP.json","view_paper":"https://pith.science/paper/PNAGIFK6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2206.02915&json=true","fetch_graph":"https://pith.science/api/pith-number/PNAGIFK6FBFCOXJYS7EOJKGTZP/graph.json","fetch_events":"https://pith.science/api/pith-number/PNAGIFK6FBFCOXJYS7EOJKGTZP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PNAGIFK6FBFCOXJYS7EOJKGTZP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PNAGIFK6FBFCOXJYS7EOJKGTZP/action/storage_attestation","attest_author":"https://pith.science/pith/PNAGIFK6FBFCOXJYS7EOJKGTZP/action/author_attestation","sign_citation":"https://pith.science/pith/PNAGIFK6FBFCOXJYS7EOJKGTZP/action/citation_signature","submit_replication":"https://pith.science/pith/PNAGIFK6FBFCOXJYS7EOJKGTZP/action/replication_record"}},"created_at":"2026-07-05T04:29:37.115678+00:00","updated_at":"2026-07-05T04:29:37.115678+00:00"}