{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:QNBNPQG4HFAM2YJSEXPJBJ5P4H","short_pith_number":"pith:QNBNPQG4","schema_version":"1.0","canonical_sha256":"8342d7c0dc3940cd613225de90a7afe1f7089b736378cc482946d787a6ff8d26","source":{"kind":"arxiv","id":"2408.14267","version":1},"attestation_state":"computed","paper":{"title":"1-Bit FQT: Pushing the Limit of Fully Quantized Training to 1-bit","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.LG","authors_text":"Chang Gao, Jianfei Chen, Jiaqi Wang, Kang Zhao, Liping Jing","submitted_at":"2024-08-26T13:42:43Z","abstract_excerpt":"Fully quantized training (FQT) accelerates the training of deep neural networks by quantizing the activations, weights, and gradients into lower precision. To explore the ultimate limit of FQT (the lowest achievable precision), we make a first attempt to 1-bit FQT. We provide a theoretical analysis of FQT based on Adam and SGD, revealing that the gradient variance influences the convergence of FQT. Building on these theoretical results, we introduce an Activation Gradient Pruning (AGP) strategy. The strategy leverages the heterogeneity of gradients by pruning less informative gradients and enh"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.14267","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-08-26T13:42:43Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"a917f621fb56023131e4cff2ff9759899b932b2b8a3cc46c6b346c3fd4ae2b26","abstract_canon_sha256":"155387dfc477749721b335f79a02ddac2a698317e649f4b3afc4a9c6b40030bc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:59:18.587185Z","signature_b64":"GAUXff14gRG2SPBZYEOraFemG/cCX2y35+/VTpuu/k23vWHUqIdX7xYm3lVw1ZFyhM5/czUq9F+lRGLmP9mEBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8342d7c0dc3940cd613225de90a7afe1f7089b736378cc482946d787a6ff8d26","last_reissued_at":"2026-07-05T08:59:18.586663Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:59:18.586663Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"1-Bit FQT: Pushing the Limit of Fully Quantized Training to 1-bit","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.LG","authors_text":"Chang Gao, Jianfei Chen, Jiaqi Wang, Kang Zhao, Liping Jing","submitted_at":"2024-08-26T13:42:43Z","abstract_excerpt":"Fully quantized training (FQT) accelerates the training of deep neural networks by quantizing the activations, weights, and gradients into lower precision. To explore the ultimate limit of FQT (the lowest achievable precision), we make a first attempt to 1-bit FQT. We provide a theoretical analysis of FQT based on Adam and SGD, revealing that the gradient variance influences the convergence of FQT. Building on these theoretical results, we introduce an Activation Gradient Pruning (AGP) strategy. The strategy leverages the heterogeneity of gradients by pruning less informative gradients and enh"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.14267","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.14267/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.14267","created_at":"2026-07-05T08:59:18.586724+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.14267v1","created_at":"2026-07-05T08:59:18.586724+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.14267","created_at":"2026-07-05T08:59:18.586724+00:00"},{"alias_kind":"pith_short_12","alias_value":"QNBNPQG4HFAM","created_at":"2026-07-05T08:59:18.586724+00:00"},{"alias_kind":"pith_short_16","alias_value":"QNBNPQG4HFAM2YJS","created_at":"2026-07-05T08:59:18.586724+00:00"},{"alias_kind":"pith_short_8","alias_value":"QNBNPQG4","created_at":"2026-07-05T08:59:18.586724+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.01043","citing_title":"Low-Precision Training of Large Language Models: Methods, Challenges, and Opportunities","ref_index":83,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QNBNPQG4HFAM2YJSEXPJBJ5P4H","json":"https://pith.science/pith/QNBNPQG4HFAM2YJSEXPJBJ5P4H.json","graph_json":"https://pith.science/api/pith-number/QNBNPQG4HFAM2YJSEXPJBJ5P4H/graph.json","events_json":"https://pith.science/api/pith-number/QNBNPQG4HFAM2YJSEXPJBJ5P4H/events.json","paper":"https://pith.science/paper/QNBNPQG4"},"agent_actions":{"view_html":"https://pith.science/pith/QNBNPQG4HFAM2YJSEXPJBJ5P4H","download_json":"https://pith.science/pith/QNBNPQG4HFAM2YJSEXPJBJ5P4H.json","view_paper":"https://pith.science/paper/QNBNPQG4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.14267&json=true","fetch_graph":"https://pith.science/api/pith-number/QNBNPQG4HFAM2YJSEXPJBJ5P4H/graph.json","fetch_events":"https://pith.science/api/pith-number/QNBNPQG4HFAM2YJSEXPJBJ5P4H/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QNBNPQG4HFAM2YJSEXPJBJ5P4H/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QNBNPQG4HFAM2YJSEXPJBJ5P4H/action/storage_attestation","attest_author":"https://pith.science/pith/QNBNPQG4HFAM2YJSEXPJBJ5P4H/action/author_attestation","sign_citation":"https://pith.science/pith/QNBNPQG4HFAM2YJSEXPJBJ5P4H/action/citation_signature","submit_replication":"https://pith.science/pith/QNBNPQG4HFAM2YJSEXPJBJ5P4H/action/replication_record"}},"created_at":"2026-07-05T08:59:18.586724+00:00","updated_at":"2026-07-05T08:59:18.586724+00:00"}