{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:C2FKGQK7K34ZUTRTICVVSLUBZQ","short_pith_number":"pith:C2FKGQK7","schema_version":"1.0","canonical_sha256":"168aa3415f56f99a4e3340ab592e81cc0d70453c92bea4dd8f8992a0c7498f6f","source":{"kind":"arxiv","id":"2403.01241","version":2},"attestation_state":"computed","paper":{"title":"IntactKV: Improving Large Language Model Quantization by Keeping Pivot Tokens Intact","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Chun Yuan, Han Gao, Haokun Lin, Haoli Bai, Jun Yao, Lu Hou, Ruikang Liu, Yuening Li, Zhengzhuo Xu","submitted_at":"2024-03-02T16:05:26Z","abstract_excerpt":"Large language models (LLMs) excel in natural language processing but demand intensive computation. To mitigate this, various quantization methods have been explored, yet they compromise LLM performance. This paper unveils a previously overlooked type of outliers in LLMs. Such outliers are found to allocate most of the attention scores on initial tokens of input, termed as pivot tokens, which are crucial to the performance of quantized LLMs. Given that, we propose IntactKV to generate the KV cache of pivot tokens losslessly from the full-precision model. The approach is simple and easy to comb"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.01241","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-03-02T16:05:26Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"a16200864baf6f712907cebb4c14955ca47ecbac5493f584a8d2b4d1a03dfbd2","abstract_canon_sha256":"28af34fdecb3dca0affdddc23935a096f3c5e963f70ec433c814572fd24c34f5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:23:19.293500Z","signature_b64":"c6XjVBuGPi/5omQSyd69vDOP75vbbwtHMTbJK5BWKGLRTBdiq3xtzCaBMut9FBdu2lZwAEjDqXPs/IlcDP6mAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"168aa3415f56f99a4e3340ab592e81cc0d70453c92bea4dd8f8992a0c7498f6f","last_reissued_at":"2026-07-05T08:23:19.293006Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:23:19.293006Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"IntactKV: Improving Large Language Model Quantization by Keeping Pivot Tokens Intact","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Chun Yuan, Han Gao, Haokun Lin, Haoli Bai, Jun Yao, Lu Hou, Ruikang Liu, Yuening Li, Zhengzhuo Xu","submitted_at":"2024-03-02T16:05:26Z","abstract_excerpt":"Large language models (LLMs) excel in natural language processing but demand intensive computation. To mitigate this, various quantization methods have been explored, yet they compromise LLM performance. This paper unveils a previously overlooked type of outliers in LLMs. Such outliers are found to allocate most of the attention scores on initial tokens of input, termed as pivot tokens, which are crucial to the performance of quantized LLMs. Given that, we propose IntactKV to generate the KV cache of pivot tokens losslessly from the full-precision model. The approach is simple and easy to comb"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.01241","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.01241/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.01241","created_at":"2026-07-05T08:23:19.293064+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.01241v2","created_at":"2026-07-05T08:23:19.293064+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.01241","created_at":"2026-07-05T08:23:19.293064+00:00"},{"alias_kind":"pith_short_12","alias_value":"C2FKGQK7K34Z","created_at":"2026-07-05T08:23:19.293064+00:00"},{"alias_kind":"pith_short_16","alias_value":"C2FKGQK7K34ZUTRT","created_at":"2026-07-05T08:23:19.293064+00:00"},{"alias_kind":"pith_short_8","alias_value":"C2FKGQK7","created_at":"2026-07-05T08:23:19.293064+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2410.10781","citing_title":"When Attention Sink Emerges in Language Models: An Empirical View","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2602.01203","citing_title":"Attention Sink Forges Native MoE in Attention Layers: Sink-Aware Training to Address Head Collapse","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06611","citing_title":"The Structural Origin of Attention Sink: Variance Discrepancy, Super Neurons, and Dimension Disparity","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/C2FKGQK7K34ZUTRTICVVSLUBZQ","json":"https://pith.science/pith/C2FKGQK7K34ZUTRTICVVSLUBZQ.json","graph_json":"https://pith.science/api/pith-number/C2FKGQK7K34ZUTRTICVVSLUBZQ/graph.json","events_json":"https://pith.science/api/pith-number/C2FKGQK7K34ZUTRTICVVSLUBZQ/events.json","paper":"https://pith.science/paper/C2FKGQK7"},"agent_actions":{"view_html":"https://pith.science/pith/C2FKGQK7K34ZUTRTICVVSLUBZQ","download_json":"https://pith.science/pith/C2FKGQK7K34ZUTRTICVVSLUBZQ.json","view_paper":"https://pith.science/paper/C2FKGQK7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.01241&json=true","fetch_graph":"https://pith.science/api/pith-number/C2FKGQK7K34ZUTRTICVVSLUBZQ/graph.json","fetch_events":"https://pith.science/api/pith-number/C2FKGQK7K34ZUTRTICVVSLUBZQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/C2FKGQK7K34ZUTRTICVVSLUBZQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/C2FKGQK7K34ZUTRTICVVSLUBZQ/action/storage_attestation","attest_author":"https://pith.science/pith/C2FKGQK7K34ZUTRTICVVSLUBZQ/action/author_attestation","sign_citation":"https://pith.science/pith/C2FKGQK7K34ZUTRTICVVSLUBZQ/action/citation_signature","submit_replication":"https://pith.science/pith/C2FKGQK7K34ZUTRTICVVSLUBZQ/action/replication_record"}},"created_at":"2026-07-05T08:23:19.293064+00:00","updated_at":"2026-07-05T08:23:19.293064+00:00"}