{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:H3DTJI6QZJXF6TLKK6ENNQKV3R","short_pith_number":"pith:H3DTJI6Q","schema_version":"1.0","canonical_sha256":"3ec734a3d0ca6e5f4d6a5788d6c155dc580cb49524dd7185e2aaf1ca42a59e16","source":{"kind":"arxiv","id":"2309.14592","version":2},"attestation_state":"computed","paper":{"title":"Efficient Post-training Quantization with FP8 Formats","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Chang Wang, Haihao Shen, Mengni Wang, Naveen Mellempudi, Qun Gao, Xin He","submitted_at":"2023-09-26T00:58:36Z","abstract_excerpt":"Recent advances in deep learning methods such as LLMs and Diffusion models have created a need for improved quantization methods that can meet the computational demands of these modern architectures while maintaining accuracy. Towards this goal, we study the advantages of FP8 data formats for post-training quantization across 75 unique network architectures covering a wide range of tasks, including machine translation, language modeling, text generation, image classification, generation, and segmentation. We examine three different FP8 representations (E5M2, E4M3, and E3M4) to study the effect"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.14592","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-09-26T00:58:36Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"028771db2c6935c45069fc05f1c6d4af33f9cc13b584f8582dbcb6cd4a913403","abstract_canon_sha256":"f1f6b7d7d61486705795847ec0a086aa0c0e71b6ee55dc94ced4a047147f5458"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:02:46.312571Z","signature_b64":"0p9pERKwd+RdazZESYVj0IYg7xubjuvKnNKnZ+puZCLkRVN87PRilOGmaOqO7GKcGO6PxRBmGMDWglQrK+kPCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3ec734a3d0ca6e5f4d6a5788d6c155dc580cb49524dd7185e2aaf1ca42a59e16","last_reissued_at":"2026-07-05T08:02:46.312091Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:02:46.312091Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Efficient Post-training Quantization with FP8 Formats","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Chang Wang, Haihao Shen, Mengni Wang, Naveen Mellempudi, Qun Gao, Xin He","submitted_at":"2023-09-26T00:58:36Z","abstract_excerpt":"Recent advances in deep learning methods such as LLMs and Diffusion models have created a need for improved quantization methods that can meet the computational demands of these modern architectures while maintaining accuracy. Towards this goal, we study the advantages of FP8 data formats for post-training quantization across 75 unique network architectures covering a wide range of tasks, including machine translation, language modeling, text generation, image classification, generation, and segmentation. We examine three different FP8 representations (E5M2, E4M3, and E3M4) to study the effect"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.14592","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.14592/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.14592","created_at":"2026-07-05T08:02:46.312153+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.14592v2","created_at":"2026-07-05T08:02:46.312153+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.14592","created_at":"2026-07-05T08:02:46.312153+00:00"},{"alias_kind":"pith_short_12","alias_value":"H3DTJI6QZJXF","created_at":"2026-07-05T08:02:46.312153+00:00"},{"alias_kind":"pith_short_16","alias_value":"H3DTJI6QZJXF6TLK","created_at":"2026-07-05T08:02:46.312153+00:00"},{"alias_kind":"pith_short_8","alias_value":"H3DTJI6Q","created_at":"2026-07-05T08:02:46.312153+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.26632","citing_title":"RT-Lynx: Putting GEMM Sparsity in the Right Place for Diffusion Models","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2601.14910","citing_title":"PipeWeave: Synergizing Analytical and Learning Models for Unified GPU Performance Prediction","ref_index":63,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/H3DTJI6QZJXF6TLKK6ENNQKV3R","json":"https://pith.science/pith/H3DTJI6QZJXF6TLKK6ENNQKV3R.json","graph_json":"https://pith.science/api/pith-number/H3DTJI6QZJXF6TLKK6ENNQKV3R/graph.json","events_json":"https://pith.science/api/pith-number/H3DTJI6QZJXF6TLKK6ENNQKV3R/events.json","paper":"https://pith.science/paper/H3DTJI6Q"},"agent_actions":{"view_html":"https://pith.science/pith/H3DTJI6QZJXF6TLKK6ENNQKV3R","download_json":"https://pith.science/pith/H3DTJI6QZJXF6TLKK6ENNQKV3R.json","view_paper":"https://pith.science/paper/H3DTJI6Q","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.14592&json=true","fetch_graph":"https://pith.science/api/pith-number/H3DTJI6QZJXF6TLKK6ENNQKV3R/graph.json","fetch_events":"https://pith.science/api/pith-number/H3DTJI6QZJXF6TLKK6ENNQKV3R/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/H3DTJI6QZJXF6TLKK6ENNQKV3R/action/timestamp_anchor","attest_storage":"https://pith.science/pith/H3DTJI6QZJXF6TLKK6ENNQKV3R/action/storage_attestation","attest_author":"https://pith.science/pith/H3DTJI6QZJXF6TLKK6ENNQKV3R/action/author_attestation","sign_citation":"https://pith.science/pith/H3DTJI6QZJXF6TLKK6ENNQKV3R/action/citation_signature","submit_replication":"https://pith.science/pith/H3DTJI6QZJXF6TLKK6ENNQKV3R/action/replication_record"}},"created_at":"2026-07-05T08:02:46.312153+00:00","updated_at":"2026-07-05T08:02:46.312153+00:00"}