{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:WXRRO67OT5Q35AXYFAMPMQWC3U","short_pith_number":"pith:WXRRO67O","schema_version":"1.0","canonical_sha256":"b5e3177bee9f61be82f82818f642c2dd1058ae15114354088a435b5ce303ebf6","source":{"kind":"arxiv","id":"2405.14428","version":1},"attestation_state":"computed","paper":{"title":"Mitigating Quantization Errors Due to Activation Spikes in GLU-Based LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Hayun Kim, Jaewoo Yang, Younghoon Kim","submitted_at":"2024-05-23T10:54:14Z","abstract_excerpt":"Modern large language models (LLMs) have established state-of-the-art performance through architectural improvements, but still require significant computational cost for inference. In an effort to reduce the inference cost, post-training quantization (PTQ) has become a popular approach, quantizing weights and activations to lower precision, such as INT8. In this paper, we reveal the challenges of activation quantization in GLU variants, which are widely used in feed-forward network (FFN) of modern LLMs, such as LLaMA family. The problem is that severe local quantization errors, caused by exce"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.14428","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-05-23T10:54:14Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"52108b06ba693e70932948c775de8948f5cb36462b60ce5e0594905a2ef4d820","abstract_canon_sha256":"e15386c48949507ec233ca5ebf1b27d7f5afa5872de617214f6e28072b634a7d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:22:14.446934Z","signature_b64":"bc3vu7nriv4a0SUIYScvvgy4bkgR+utVpK8aiY6EMCOqaC4Ocmdr9ys4Ju6Pt+aGMzVFqeRZp4YDliXT6QOACQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b5e3177bee9f61be82f82818f642c2dd1058ae15114354088a435b5ce303ebf6","last_reissued_at":"2026-07-05T08:22:14.446602Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:22:14.446602Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Mitigating Quantization Errors Due to Activation Spikes in GLU-Based LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Hayun Kim, Jaewoo Yang, Younghoon Kim","submitted_at":"2024-05-23T10:54:14Z","abstract_excerpt":"Modern large language models (LLMs) have established state-of-the-art performance through architectural improvements, but still require significant computational cost for inference. In an effort to reduce the inference cost, post-training quantization (PTQ) has become a popular approach, quantizing weights and activations to lower precision, such as INT8. In this paper, we reveal the challenges of activation quantization in GLU variants, which are widely used in feed-forward network (FFN) of modern LLMs, such as LLaMA family. The problem is that severe local quantization errors, caused by exce"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.14428","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.14428/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.14428","created_at":"2026-07-05T08:22:14.446651+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.14428v1","created_at":"2026-07-05T08:22:14.446651+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.14428","created_at":"2026-07-05T08:22:14.446651+00:00"},{"alias_kind":"pith_short_12","alias_value":"WXRRO67OT5Q3","created_at":"2026-07-05T08:22:14.446651+00:00"},{"alias_kind":"pith_short_16","alias_value":"WXRRO67OT5Q35AXY","created_at":"2026-07-05T08:22:14.446651+00:00"},{"alias_kind":"pith_short_8","alias_value":"WXRRO67O","created_at":"2026-07-05T08:22:14.446651+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01607","citing_title":"MxGLUT: A Reconfigurable LUT-Centric Broadcast Dataflow Accelerator for Mixed-Precision GEMM","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02288","citing_title":"Massive Spikes in LLMs are Bias Vectors: Mechanistic Uncovering and Spike-Free Quantization","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23191","citing_title":"Expand More, Shrink Less: Shaping Effective-Rank Dynamics for Dense Scaling in Recommendation","ref_index":40,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WXRRO67OT5Q35AXYFAMPMQWC3U","json":"https://pith.science/pith/WXRRO67OT5Q35AXYFAMPMQWC3U.json","graph_json":"https://pith.science/api/pith-number/WXRRO67OT5Q35AXYFAMPMQWC3U/graph.json","events_json":"https://pith.science/api/pith-number/WXRRO67OT5Q35AXYFAMPMQWC3U/events.json","paper":"https://pith.science/paper/WXRRO67O"},"agent_actions":{"view_html":"https://pith.science/pith/WXRRO67OT5Q35AXYFAMPMQWC3U","download_json":"https://pith.science/pith/WXRRO67OT5Q35AXYFAMPMQWC3U.json","view_paper":"https://pith.science/paper/WXRRO67O","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.14428&json=true","fetch_graph":"https://pith.science/api/pith-number/WXRRO67OT5Q35AXYFAMPMQWC3U/graph.json","fetch_events":"https://pith.science/api/pith-number/WXRRO67OT5Q35AXYFAMPMQWC3U/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WXRRO67OT5Q35AXYFAMPMQWC3U/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WXRRO67OT5Q35AXYFAMPMQWC3U/action/storage_attestation","attest_author":"https://pith.science/pith/WXRRO67OT5Q35AXYFAMPMQWC3U/action/author_attestation","sign_citation":"https://pith.science/pith/WXRRO67OT5Q35AXYFAMPMQWC3U/action/citation_signature","submit_replication":"https://pith.science/pith/WXRRO67OT5Q35AXYFAMPMQWC3U/action/replication_record"}},"created_at":"2026-07-05T08:22:14.446651+00:00","updated_at":"2026-07-05T08:22:14.446651+00:00"}