{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:4CAQC7FG2DHKQXT2SYNXAYYPHZ","short_pith_number":"pith:4CAQC7FG","schema_version":"1.0","canonical_sha256":"e081017ca6d0cea85e7a961b70630f3e639cffb946a821375ee7b24175049303","source":{"kind":"arxiv","id":"2004.09602","version":1},"attestation_state":"computed","paper":{"title":"Integer Quantization for Deep Learning Inference: Principles and Empirical Evaluation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Hao Wu, Mikhail Isaev, Patrick Judd, Paulius Micikevicius, Xiaojie Zhang","submitted_at":"2020-04-20T19:59:22Z","abstract_excerpt":"Quantization techniques can reduce the size of Deep Neural Networks and improve inference latency and throughput by taking advantage of high throughput integer instructions. In this paper we review the mathematical aspects of quantization parameters and evaluate their choices on a wide range of neural network models for different application domains, including vision, speech, and language. We focus on quantization techniques that are amenable to acceleration by processors with high-throughput integer math pipelines. We also present a workflow for 8-bit quantization that is able to maintain acc"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2004.09602","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-04-20T19:59:22Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"867e469bc05945838cdf93d7f7f711f3de66cfbe49619d4f81e4956eef96a7e4","abstract_canon_sha256":"bf0fe0e6a74ff3b8eb123609680bc64814cf4f63ad86e6ed3b32c6dd2a6c3d60"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:56:51.904177Z","signature_b64":"aSW0SX5XSkbyefiqtV4Hxgk7vTZw3KYnNWrIVQtWCLTevmzV9Q9ZDMuX/oVFH50drmszc8SJhDt5c1bpiEu1AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e081017ca6d0cea85e7a961b70630f3e639cffb946a821375ee7b24175049303","last_reissued_at":"2026-07-05T00:56:51.903702Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:56:51.903702Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Integer Quantization for Deep Learning Inference: Principles and Empirical Evaluation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Hao Wu, Mikhail Isaev, Patrick Judd, Paulius Micikevicius, Xiaojie Zhang","submitted_at":"2020-04-20T19:59:22Z","abstract_excerpt":"Quantization techniques can reduce the size of Deep Neural Networks and improve inference latency and throughput by taking advantage of high throughput integer instructions. In this paper we review the mathematical aspects of quantization parameters and evaluate their choices on a wide range of neural network models for different application domains, including vision, speech, and language. We focus on quantization techniques that are amenable to acceleration by processors with high-throughput integer math pipelines. We also present a workflow for 8-bit quantization that is able to maintain acc"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2004.09602","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2004.09602/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2004.09602","created_at":"2026-07-05T00:56:51.903756+00:00"},{"alias_kind":"arxiv_version","alias_value":"2004.09602v1","created_at":"2026-07-05T00:56:51.903756+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2004.09602","created_at":"2026-07-05T00:56:51.903756+00:00"},{"alias_kind":"pith_short_12","alias_value":"4CAQC7FG2DHK","created_at":"2026-07-05T00:56:51.903756+00:00"},{"alias_kind":"pith_short_16","alias_value":"4CAQC7FG2DHKQXT2","created_at":"2026-07-05T00:56:51.903756+00:00"},{"alias_kind":"pith_short_8","alias_value":"4CAQC7FG","created_at":"2026-07-05T00:56:51.903756+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20937","citing_title":"Learning through Internalization","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28600","citing_title":"Transformers Provably Learn to Internalize Chain-of-Thought","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22351","citing_title":"QuantSR+: Pushing the Limit of Quantized Image Super-Resolution Networks","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2209.05433","citing_title":"FP8 Formats for Deep Learning","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2208.07339","citing_title":"LLM.int8(): 8-bit Matrix Multiplication for Transformers at Scale","ref_index":170,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10959","citing_title":"QuIDE: Mastering the Quantized Intelligence Trade-off via Active Optimization","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26857","citing_title":"Edge AI for Automotive Vulnerable Road User Safety: Deployable Detection via Knowledge Distillation","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26505","citing_title":"Quantamination: Dynamic Quantization Leaks Your Data Across the Batch","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23426","citing_title":"Enhanced Privacy and Communication Efficiency in Non-IID Federated Learning with Adaptive Quantization and Differential Privacy","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14314","citing_title":"DharmaOCR: Specialized Small Language Models for Structured OCR that outperform Open-Source and Commercial Baselines","ref_index":56,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4CAQC7FG2DHKQXT2SYNXAYYPHZ","json":"https://pith.science/pith/4CAQC7FG2DHKQXT2SYNXAYYPHZ.json","graph_json":"https://pith.science/api/pith-number/4CAQC7FG2DHKQXT2SYNXAYYPHZ/graph.json","events_json":"https://pith.science/api/pith-number/4CAQC7FG2DHKQXT2SYNXAYYPHZ/events.json","paper":"https://pith.science/paper/4CAQC7FG"},"agent_actions":{"view_html":"https://pith.science/pith/4CAQC7FG2DHKQXT2SYNXAYYPHZ","download_json":"https://pith.science/pith/4CAQC7FG2DHKQXT2SYNXAYYPHZ.json","view_paper":"https://pith.science/paper/4CAQC7FG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2004.09602&json=true","fetch_graph":"https://pith.science/api/pith-number/4CAQC7FG2DHKQXT2SYNXAYYPHZ/graph.json","fetch_events":"https://pith.science/api/pith-number/4CAQC7FG2DHKQXT2SYNXAYYPHZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4CAQC7FG2DHKQXT2SYNXAYYPHZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4CAQC7FG2DHKQXT2SYNXAYYPHZ/action/storage_attestation","attest_author":"https://pith.science/pith/4CAQC7FG2DHKQXT2SYNXAYYPHZ/action/author_attestation","sign_citation":"https://pith.science/pith/4CAQC7FG2DHKQXT2SYNXAYYPHZ/action/citation_signature","submit_replication":"https://pith.science/pith/4CAQC7FG2DHKQXT2SYNXAYYPHZ/action/replication_record"}},"created_at":"2026-07-05T00:56:51.903756+00:00","updated_at":"2026-07-05T00:56:51.903756+00:00"}