{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:I7EEFQHPZKN467RNZN3WDCL2P2","short_pith_number":"pith:I7EEFQHP","schema_version":"1.0","canonical_sha256":"47c842c0efca9bcf7e2dcb7761897a7e8ba8312df8671701cd3dc2180b092ed1","source":{"kind":"arxiv","id":"2406.06385","version":3},"attestation_state":"computed","paper":{"title":"Low-Rank Quantization-Aware Training for LLMs","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Markus Nagel, Riccardo Del Chiaro, Yelysei Bondarenko","submitted_at":"2024-06-10T15:44:22Z","abstract_excerpt":"Large language models (LLMs) are omnipresent, however their practical deployment is challenging due to their ever increasing computational and memory demands. Quantization is one of the most effective ways to make them more compute and memory efficient. Quantization-aware training (QAT) methods, generally produce the best quantized performance, however it comes at the cost of potentially long training time and excessive memory usage, making it impractical when applying for LLMs. Inspired by parameter-efficient fine-tuning (PEFT) and low-rank adaptation (LoRA) literature, we propose LR-QAT -- a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.06385","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2024-06-10T15:44:22Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"92fc2f47ba16b0e955393e4dc674cb0f8767aac9d46b814d31c708a3cb2cf21f","abstract_canon_sha256":"61f15fa527e59f8cd9dd06b19406e17d83dd44c7544a29b0082a81c29f0364a6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:02:21.073286Z","signature_b64":"p+Ct5sVxch3zEdrvWCkWqzJK5Wh/rygHCIJDZaAtMt9DMZMKEuPa+XKPBUawOF+NJNir3ratepWonjjadMcbDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"47c842c0efca9bcf7e2dcb7761897a7e8ba8312df8671701cd3dc2180b092ed1","last_reissued_at":"2026-07-05T09:02:21.072810Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:02:21.072810Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Low-Rank Quantization-Aware Training for LLMs","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Markus Nagel, Riccardo Del Chiaro, Yelysei Bondarenko","submitted_at":"2024-06-10T15:44:22Z","abstract_excerpt":"Large language models (LLMs) are omnipresent, however their practical deployment is challenging due to their ever increasing computational and memory demands. Quantization is one of the most effective ways to make them more compute and memory efficient. Quantization-aware training (QAT) methods, generally produce the best quantized performance, however it comes at the cost of potentially long training time and excessive memory usage, making it impractical when applying for LLMs. Inspired by parameter-efficient fine-tuning (PEFT) and low-rank adaptation (LoRA) literature, we propose LR-QAT -- a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.06385","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.06385/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.06385","created_at":"2026-07-05T09:02:21.072867+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.06385v3","created_at":"2026-07-05T09:02:21.072867+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.06385","created_at":"2026-07-05T09:02:21.072867+00:00"},{"alias_kind":"pith_short_12","alias_value":"I7EEFQHPZKN4","created_at":"2026-07-05T09:02:21.072867+00:00"},{"alias_kind":"pith_short_16","alias_value":"I7EEFQHPZKN467RN","created_at":"2026-07-05T09:02:21.072867+00:00"},{"alias_kind":"pith_short_8","alias_value":"I7EEFQHP","created_at":"2026-07-05T09:02:21.072867+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.25966","citing_title":"Mapping the Schedule x Bit-Width Boundary in Sub-100M Quantisation-Aware Training","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00539","citing_title":"GNMR: Runtime Stability Control for Low-Precision Large Language Model Training","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2509.21173","citing_title":"Less Precise Can Be More Reliable: A Systematic Evaluation of Quantization's Impact on VLMs Beyond Accuracy","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20295","citing_title":"Quant.npu: Enabling Efficient Mobile NPU Inference for on-device LLMs via Fully Static Quantization","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19929","citing_title":"Breaking Modality Heterogeneity in Low-Bit Quantization for Large Vision-Language Models","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2509.21173","citing_title":"Less Precise Can Be More Reliable: A Systematic Evaluation of Quantization's Impact on VLMs Beyond Accuracy","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10673","citing_title":"Compander-Aligned Query Geometry for Quantized Zeroth-Order Optimization","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02726","citing_title":"Cool-chic 5.0: Faster Encoding and Inter-Feature Entropy Modeling for Overfitted Image Compression","ref_index":47,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/I7EEFQHPZKN467RNZN3WDCL2P2","json":"https://pith.science/pith/I7EEFQHPZKN467RNZN3WDCL2P2.json","graph_json":"https://pith.science/api/pith-number/I7EEFQHPZKN467RNZN3WDCL2P2/graph.json","events_json":"https://pith.science/api/pith-number/I7EEFQHPZKN467RNZN3WDCL2P2/events.json","paper":"https://pith.science/paper/I7EEFQHP"},"agent_actions":{"view_html":"https://pith.science/pith/I7EEFQHPZKN467RNZN3WDCL2P2","download_json":"https://pith.science/pith/I7EEFQHPZKN467RNZN3WDCL2P2.json","view_paper":"https://pith.science/paper/I7EEFQHP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.06385&json=true","fetch_graph":"https://pith.science/api/pith-number/I7EEFQHPZKN467RNZN3WDCL2P2/graph.json","fetch_events":"https://pith.science/api/pith-number/I7EEFQHPZKN467RNZN3WDCL2P2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/I7EEFQHPZKN467RNZN3WDCL2P2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/I7EEFQHPZKN467RNZN3WDCL2P2/action/storage_attestation","attest_author":"https://pith.science/pith/I7EEFQHPZKN467RNZN3WDCL2P2/action/author_attestation","sign_citation":"https://pith.science/pith/I7EEFQHPZKN467RNZN3WDCL2P2/action/citation_signature","submit_replication":"https://pith.science/pith/I7EEFQHPZKN467RNZN3WDCL2P2/action/replication_record"}},"created_at":"2026-07-05T09:02:21.072867+00:00","updated_at":"2026-07-05T09:02:21.072867+00:00"}