{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:55MRM4TBUS7FCL224HW7EWR747","short_pith_number":"pith:55MRM4TB","schema_version":"1.0","canonical_sha256":"ef59167261a4be512f5ae1edf25a3fe7e72359f75d29bcebd8e6f77fb1d3c392","source":{"kind":"arxiv","id":"2508.04257","version":1},"attestation_state":"computed","paper":{"title":"KVSink: Understanding and Enhancing the Preservation of Attention Sinks in KV Cache Quantization for LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Kehong Yuan, Zunhai Su","submitted_at":"2025-08-06T09:40:09Z","abstract_excerpt":"Key-Value (KV) cache quantization has become a widely adopted optimization technique for efficient large language models (LLMs) inference by reducing KV cache memory usage and mitigating memory-bound constraints. Recent studies have emphasized the importance of preserving the original precision of KVs for the first few tokens to ensure the protection of attention sinks. While this approach has proven effective in mitigating performance degradation, its underlying principles remain insufficiently understood. Moreover, it fails to address the recent discovery that attention sinks can emerge beyo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.04257","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-08-06T09:40:09Z","cross_cats_sorted":[],"title_canon_sha256":"2d37615389084c5e882ea22d773cf075ba1b0d0fd49a54c31db2e473eea19ed9","abstract_canon_sha256":"a99a6aed3b7f8f9b646e6a95ae6fdfaaeedda1116d70abd48c3cb0833a04cda2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:49:23.768885Z","signature_b64":"4zziyvIwgPhNLLq3goeHWAmMj1mu86/GPtykI6R8LEs3p0TCr+mYq2eDHBqpD+4Ht0bWFpW/SEyS1dyOaPU8CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ef59167261a4be512f5ae1edf25a3fe7e72359f75d29bcebd8e6f77fb1d3c392","last_reissued_at":"2026-07-05T11:49:23.768429Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:49:23.768429Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"KVSink: Understanding and Enhancing the Preservation of Attention Sinks in KV Cache Quantization for LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Kehong Yuan, Zunhai Su","submitted_at":"2025-08-06T09:40:09Z","abstract_excerpt":"Key-Value (KV) cache quantization has become a widely adopted optimization technique for efficient large language models (LLMs) inference by reducing KV cache memory usage and mitigating memory-bound constraints. Recent studies have emphasized the importance of preserving the original precision of KVs for the first few tokens to ensure the protection of attention sinks. While this approach has proven effective in mitigating performance degradation, its underlying principles remain insufficiently understood. Moreover, it fails to address the recent discovery that attention sinks can emerge beyo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.04257","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.04257/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.04257","created_at":"2026-07-05T11:49:23.768486+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.04257v1","created_at":"2026-07-05T11:49:23.768486+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.04257","created_at":"2026-07-05T11:49:23.768486+00:00"},{"alias_kind":"pith_short_12","alias_value":"55MRM4TBUS7F","created_at":"2026-07-05T11:49:23.768486+00:00"},{"alias_kind":"pith_short_16","alias_value":"55MRM4TBUS7FCL22","created_at":"2026-07-05T11:49:23.768486+00:00"},{"alias_kind":"pith_short_8","alias_value":"55MRM4TB","created_at":"2026-07-05T11:49:23.768486+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.25244","citing_title":"Inference Time Optimization with Confidence Dynamics","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19660","citing_title":"OScaR: The Occam's Razor for Extreme KV Cache Quantization in LLMs and Beyond","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2601.14004","citing_title":"Locate, Steer, and Improve: A Practical Survey of Actionable Mechanistic Interpretability in Large Language Models","ref_index":286,"is_internal_anchor":false},{"citing_arxiv_id":"2602.10718","citing_title":"SnapMLA: Efficient Long-Context MLA Decoding via Hardware-Aware FP8 Quantized Pipelining","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03316","citing_title":"When Sinks Help or Hurt: Unified Framework for Attention Sink in Large Vision-Language Models","ref_index":35,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/55MRM4TBUS7FCL224HW7EWR747","json":"https://pith.science/pith/55MRM4TBUS7FCL224HW7EWR747.json","graph_json":"https://pith.science/api/pith-number/55MRM4TBUS7FCL224HW7EWR747/graph.json","events_json":"https://pith.science/api/pith-number/55MRM4TBUS7FCL224HW7EWR747/events.json","paper":"https://pith.science/paper/55MRM4TB"},"agent_actions":{"view_html":"https://pith.science/pith/55MRM4TBUS7FCL224HW7EWR747","download_json":"https://pith.science/pith/55MRM4TBUS7FCL224HW7EWR747.json","view_paper":"https://pith.science/paper/55MRM4TB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.04257&json=true","fetch_graph":"https://pith.science/api/pith-number/55MRM4TBUS7FCL224HW7EWR747/graph.json","fetch_events":"https://pith.science/api/pith-number/55MRM4TBUS7FCL224HW7EWR747/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/55MRM4TBUS7FCL224HW7EWR747/action/timestamp_anchor","attest_storage":"https://pith.science/pith/55MRM4TBUS7FCL224HW7EWR747/action/storage_attestation","attest_author":"https://pith.science/pith/55MRM4TBUS7FCL224HW7EWR747/action/author_attestation","sign_citation":"https://pith.science/pith/55MRM4TBUS7FCL224HW7EWR747/action/citation_signature","submit_replication":"https://pith.science/pith/55MRM4TBUS7FCL224HW7EWR747/action/replication_record"}},"created_at":"2026-07-05T11:49:23.768486+00:00","updated_at":"2026-07-05T11:49:23.768486+00:00"}