{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:TOALUIUFJBV45I7AKSSU4K2GBE","short_pith_number":"pith:TOALUIUF","schema_version":"1.0","canonical_sha256":"9b80ba2285486bcea3e054a54e2b46093f2d32b98bd843b2abfa5aff72d4b0ca","source":{"kind":"arxiv","id":"2505.14638","version":1},"attestation_state":"computed","paper":{"title":"Dual Precision Quantization for Efficient and Accurate Deep Neural Networks Inference","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Asaf Karnieli, Tomer Gafni, Yair Hanani","submitted_at":"2025-05-20T17:26:12Z","abstract_excerpt":"Deep neural networks have achieved state-of-the-art results in a wide range of applications, from natural language processing and computer vision to speech recognition. However, as tasks become increasingly complex, model sizes continue to grow, posing challenges in latency and memory efficiency. To meet these constraints, post-training quantization has emerged as a promising solution. In this paper, we propose a novel hardware-efficient quantization and inference scheme that exploits hardware advantages with minimal accuracy degradation. Specifically, we introduce a W4A8 scheme, where weights"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.14638","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-05-20T17:26:12Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"0826b702f679ea55475bf2bf91584ae1cad7ad33eb054c4082b533df8f11627b","abstract_canon_sha256":"b6b4536d3e3d9aac7bb5b3b5fc02d14603d6af0b30d0da29f4cf125acd8856ea"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:06:16.849326Z","signature_b64":"Fykg8XSDFEjXPTz153gogb4mIlcFZf+AtCTHgO/zaSyMHFaXUUPRH+elq3WkvFwEU447Q6HtitW2HpQmrlhNBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9b80ba2285486bcea3e054a54e2b46093f2d32b98bd843b2abfa5aff72d4b0ca","last_reissued_at":"2026-07-05T11:06:16.848820Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:06:16.848820Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Dual Precision Quantization for Efficient and Accurate Deep Neural Networks Inference","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Asaf Karnieli, Tomer Gafni, Yair Hanani","submitted_at":"2025-05-20T17:26:12Z","abstract_excerpt":"Deep neural networks have achieved state-of-the-art results in a wide range of applications, from natural language processing and computer vision to speech recognition. However, as tasks become increasingly complex, model sizes continue to grow, posing challenges in latency and memory efficiency. To meet these constraints, post-training quantization has emerged as a promising solution. In this paper, we propose a novel hardware-efficient quantization and inference scheme that exploits hardware advantages with minimal accuracy degradation. Specifically, we introduce a W4A8 scheme, where weights"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.14638","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.14638/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.14638","created_at":"2026-07-05T11:06:16.848898+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.14638v1","created_at":"2026-07-05T11:06:16.848898+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.14638","created_at":"2026-07-05T11:06:16.848898+00:00"},{"alias_kind":"pith_short_12","alias_value":"TOALUIUFJBV4","created_at":"2026-07-05T11:06:16.848898+00:00"},{"alias_kind":"pith_short_16","alias_value":"TOALUIUFJBV45I7A","created_at":"2026-07-05T11:06:16.848898+00:00"},{"alias_kind":"pith_short_8","alias_value":"TOALUIUF","created_at":"2026-07-05T11:06:16.848898+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TOALUIUFJBV45I7AKSSU4K2GBE","json":"https://pith.science/pith/TOALUIUFJBV45I7AKSSU4K2GBE.json","graph_json":"https://pith.science/api/pith-number/TOALUIUFJBV45I7AKSSU4K2GBE/graph.json","events_json":"https://pith.science/api/pith-number/TOALUIUFJBV45I7AKSSU4K2GBE/events.json","paper":"https://pith.science/paper/TOALUIUF"},"agent_actions":{"view_html":"https://pith.science/pith/TOALUIUFJBV45I7AKSSU4K2GBE","download_json":"https://pith.science/pith/TOALUIUFJBV45I7AKSSU4K2GBE.json","view_paper":"https://pith.science/paper/TOALUIUF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.14638&json=true","fetch_graph":"https://pith.science/api/pith-number/TOALUIUFJBV45I7AKSSU4K2GBE/graph.json","fetch_events":"https://pith.science/api/pith-number/TOALUIUFJBV45I7AKSSU4K2GBE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TOALUIUFJBV45I7AKSSU4K2GBE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TOALUIUFJBV45I7AKSSU4K2GBE/action/storage_attestation","attest_author":"https://pith.science/pith/TOALUIUFJBV45I7AKSSU4K2GBE/action/author_attestation","sign_citation":"https://pith.science/pith/TOALUIUFJBV45I7AKSSU4K2GBE/action/citation_signature","submit_replication":"https://pith.science/pith/TOALUIUFJBV45I7AKSSU4K2GBE/action/replication_record"}},"created_at":"2026-07-05T11:06:16.848898+00:00","updated_at":"2026-07-05T11:06:16.848898+00:00"}