{"id":"f1e8dab3-c2cf-445a-973e-4c2abbee0851","arxiv_id":"2505.14742","paper_version":2,"verdict":"CONDITIONAL","confidence":"MODERATE","novelty_score":5.0,"correctness_risk":"medium","formal_verification":"none","parameter_count":4,"one_line_summary":"Quaff shows that activation outlier channels keep their spatial positions during LLM fine-tuning, and exploits this stability to cut fine-tuning memory and latency with INT8 quantization while matching or beating full-precision accuracy.","lead":"A new method called Quaff makes fine-tuning large language models on consumer GPUs cheaper by keeping most weights in 8-bit integers while retaining a small set of outlier channels in full precision. The paper's key hypothesis is that activation outlier channels keep stable positions during fine-tuning, which lets Quaff precompute where scaling is needed.","discovery_kind":"new_method","skeptic_critique":null,"referee_report":null,"author_rebuttal":null,"desk_editor":null,"rs_alignment":null,"lean_confirmation":null,"pith_extraction":null,"created_at":"2026-08-07T15:43:29.544810+00:00","model_set":{"reader":"deepseek-v4-flash"},"falsifier":null,"supporting_citations":[],"review_version":1}