{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:BZ4J6HSSNIBDPVRZONA5LPNQ4C","short_pith_number":"pith:BZ4J6HSS","schema_version":"1.0","canonical_sha256":"0e789f1e526a0237d6397341d5bdb0e0802a0269a72394bb58dbfa937135243a","source":{"kind":"arxiv","id":"2505.18231","version":3},"attestation_state":"computed","paper":{"title":"NSNQuant: A Double Normalization Approach for Calibration-Free Low-Bit Vector Quantization of KV Cache","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Donghyun Son, Euntae Choi, Sungjoo Yoo","submitted_at":"2025-05-23T12:40:07Z","abstract_excerpt":"Large Language Model (LLM) inference is typically memory-intensive, especially when processing large batch sizes and long sequences, due to the large size of key-value (KV) cache. Vector Quantization (VQ) is recently adopted to alleviate this issue, but we find that the existing approach is susceptible to distribution shift due to its reliance on calibration datasets. To address this limitation, we introduce NSNQuant, a calibration-free Vector Quantization (VQ) technique designed for low-bit compression of the KV cache. By applying a three-step transformation-1) a token-wise normalization (Nor"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.18231","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-23T12:40:07Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"0554b80fdb34b745e5598e379515eac2be1af30a498c28ded69f9538623b1195","abstract_canon_sha256":"5e13d7b07df271ef56109ef14a70fb23181d2e1fb6b6e12c71ed546430846c5a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-16T01:22:25.472722Z","signature_b64":"m5L1HWBoNnUkpkg2+tLnOf/rLQ1IsTCDeuWoZxEFZ82j/auxPlnqk19YNuPflss0g+u2mYnj/ZyN9p8RuU2tCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0e789f1e526a0237d6397341d5bdb0e0802a0269a72394bb58dbfa937135243a","last_reissued_at":"2026-07-16T01:22:25.471790Z","signature_status":"signed_v1","first_computed_at":"2026-07-16T01:22:25.471790Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"NSNQuant: A Double Normalization Approach for Calibration-Free Low-Bit Vector Quantization of KV Cache","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Donghyun Son, Euntae Choi, Sungjoo Yoo","submitted_at":"2025-05-23T12:40:07Z","abstract_excerpt":"Large Language Model (LLM) inference is typically memory-intensive, especially when processing large batch sizes and long sequences, due to the large size of key-value (KV) cache. Vector Quantization (VQ) is recently adopted to alleviate this issue, but we find that the existing approach is susceptible to distribution shift due to its reliance on calibration datasets. To address this limitation, we introduce NSNQuant, a calibration-free Vector Quantization (VQ) technique designed for low-bit compression of the KV cache. By applying a three-step transformation-1) a token-wise normalization (Nor"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.18231","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.18231/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.18231","created_at":"2026-07-16T01:22:25.472220+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.18231v3","created_at":"2026-07-16T01:22:25.472220+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.18231","created_at":"2026-07-16T01:22:25.472220+00:00"},{"alias_kind":"pith_short_12","alias_value":"BZ4J6HSSNIBD","created_at":"2026-07-16T01:22:25.472220+00:00"},{"alias_kind":"pith_short_16","alias_value":"BZ4J6HSSNIBDPVRZ","created_at":"2026-07-16T01:22:25.472220+00:00"},{"alias_kind":"pith_short_8","alias_value":"BZ4J6HSS","created_at":"2026-07-16T01:22:25.472220+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BZ4J6HSSNIBDPVRZONA5LPNQ4C","json":"https://pith.science/pith/BZ4J6HSSNIBDPVRZONA5LPNQ4C.json","graph_json":"https://pith.science/api/pith-number/BZ4J6HSSNIBDPVRZONA5LPNQ4C/graph.json","events_json":"https://pith.science/api/pith-number/BZ4J6HSSNIBDPVRZONA5LPNQ4C/events.json","paper":"https://pith.science/paper/BZ4J6HSS"},"agent_actions":{"view_html":"https://pith.science/pith/BZ4J6HSSNIBDPVRZONA5LPNQ4C","download_json":"https://pith.science/pith/BZ4J6HSSNIBDPVRZONA5LPNQ4C.json","view_paper":"https://pith.science/paper/BZ4J6HSS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.18231&json=true","fetch_graph":"https://pith.science/api/pith-number/BZ4J6HSSNIBDPVRZONA5LPNQ4C/graph.json","fetch_events":"https://pith.science/api/pith-number/BZ4J6HSSNIBDPVRZONA5LPNQ4C/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BZ4J6HSSNIBDPVRZONA5LPNQ4C/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BZ4J6HSSNIBDPVRZONA5LPNQ4C/action/storage_attestation","attest_author":"https://pith.science/pith/BZ4J6HSSNIBDPVRZONA5LPNQ4C/action/author_attestation","sign_citation":"https://pith.science/pith/BZ4J6HSSNIBDPVRZONA5LPNQ4C/action/citation_signature","submit_replication":"https://pith.science/pith/BZ4J6HSSNIBDPVRZONA5LPNQ4C/action/replication_record"}},"created_at":"2026-07-16T01:22:25.472220+00:00","updated_at":"2026-07-16T01:22:25.472220+00:00"}