{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:EZMF6WORHUDHROLVUD7M4ITYVG","short_pith_number":"pith:EZMF6WOR","schema_version":"1.0","canonical_sha256":"26585f59d13d0678b975a0fece2278a99b46ea51744f1f435096eaac32dd8368","source":{"kind":"arxiv","id":"2406.03482","version":2},"attestation_state":"computed","paper":{"title":"QJL: 1-Bit Quantized JL Transform for KV Cache Quantization with Zero Overhead","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.PF"],"primary_cat":"cs.LG","authors_text":"Amir Zandieh, Insu Han, Majid Daliri","submitted_at":"2024-06-05T17:42:05Z","abstract_excerpt":"Serving LLMs requires substantial memory due to the storage requirements of Key-Value (KV) embeddings in the KV cache, which grows with sequence length. An effective approach to compress KV cache is quantization. However, traditional quantization methods face significant memory overhead due to the need to store quantization constants (at least a zero point and a scale) in full precision per data block. Depending on the block size, this overhead can add 1 or 2 bits per quantized number. We introduce QJL, a new quantization approach that consists of a Johnson-Lindenstrauss (JL) transform followe"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.03482","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-06-05T17:42:05Z","cross_cats_sorted":["cs.AI","cs.CL","cs.PF"],"title_canon_sha256":"d3bd00ea3b46ecba46b9413ddc3898e86beed77fbd1dc6eba98eda4dd2df2cf2","abstract_canon_sha256":"2b090e273cc3624194b511ed4645f47102c242cdb7c9ba4bbbcddaa545471ee0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:45:29.301399Z","signature_b64":"qzTCrANZ9uCCcMf6JC7gBjUjS+liJpBFxvvR7NOVmyiGLaufR0FzzTuqoi4JrHtebJkUYOVgaY4Z5PB9dnq8Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"26585f59d13d0678b975a0fece2278a99b46ea51744f1f435096eaac32dd8368","last_reissued_at":"2026-07-05T08:45:29.300979Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:45:29.300979Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"QJL: 1-Bit Quantized JL Transform for KV Cache Quantization with Zero Overhead","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.PF"],"primary_cat":"cs.LG","authors_text":"Amir Zandieh, Insu Han, Majid Daliri","submitted_at":"2024-06-05T17:42:05Z","abstract_excerpt":"Serving LLMs requires substantial memory due to the storage requirements of Key-Value (KV) embeddings in the KV cache, which grows with sequence length. An effective approach to compress KV cache is quantization. However, traditional quantization methods face significant memory overhead due to the need to store quantization constants (at least a zero point and a scale) in full precision per data block. Depending on the block size, this overhead can add 1 or 2 bits per quantized number. We introduce QJL, a new quantization approach that consists of a Johnson-Lindenstrauss (JL) transform followe"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.03482","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.03482/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.03482","created_at":"2026-07-05T08:45:29.301033+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.03482v2","created_at":"2026-07-05T08:45:29.301033+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.03482","created_at":"2026-07-05T08:45:29.301033+00:00"},{"alias_kind":"pith_short_12","alias_value":"EZMF6WORHUDH","created_at":"2026-07-05T08:45:29.301033+00:00"},{"alias_kind":"pith_short_16","alias_value":"EZMF6WORHUDHROLV","created_at":"2026-07-05T08:45:29.301033+00:00"},{"alias_kind":"pith_short_8","alias_value":"EZMF6WOR","created_at":"2026-07-05T08:45:29.301033+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23406","citing_title":"HyperQuant: A Rate-Distortion-Optimal Quantization Pipeline for Large Language and Diffusion Models","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20474","citing_title":"UltraQuant: 4-bit KV Caching for Context-Heavy Agents","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20868","citing_title":"Runtime-Certified Bounded-Error Quantized Attention","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2504.19874","citing_title":"TurboQuant: Online Vector Quantization with Near-optimal Distortion Rate","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02638","citing_title":"AXELRAM: Quantize Once, Never Dequantize","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08114","citing_title":"Statistical Inference and Quality Measures of KV Cache Quantisations Inspired by TurboQuant","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04514","citing_title":"SuperLocalMemory V3.3: The Living Brain -- Biologically-Inspired Forgetting, Cognitive Quantization, and Multi-Channel Retrieval for Zero-LLM Agent Memory Systems","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14442","citing_title":"Hierarchical vs. Flat Iteration in Shared-Weight Transformers","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16957","citing_title":"Open-TQ-Metal: Fused Compressed-Domain Attention for Long-Context LLM Inference on Apple Silicon","ref_index":11,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EZMF6WORHUDHROLVUD7M4ITYVG","json":"https://pith.science/pith/EZMF6WORHUDHROLVUD7M4ITYVG.json","graph_json":"https://pith.science/api/pith-number/EZMF6WORHUDHROLVUD7M4ITYVG/graph.json","events_json":"https://pith.science/api/pith-number/EZMF6WORHUDHROLVUD7M4ITYVG/events.json","paper":"https://pith.science/paper/EZMF6WOR"},"agent_actions":{"view_html":"https://pith.science/pith/EZMF6WORHUDHROLVUD7M4ITYVG","download_json":"https://pith.science/pith/EZMF6WORHUDHROLVUD7M4ITYVG.json","view_paper":"https://pith.science/paper/EZMF6WOR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.03482&json=true","fetch_graph":"https://pith.science/api/pith-number/EZMF6WORHUDHROLVUD7M4ITYVG/graph.json","fetch_events":"https://pith.science/api/pith-number/EZMF6WORHUDHROLVUD7M4ITYVG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EZMF6WORHUDHROLVUD7M4ITYVG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EZMF6WORHUDHROLVUD7M4ITYVG/action/storage_attestation","attest_author":"https://pith.science/pith/EZMF6WORHUDHROLVUD7M4ITYVG/action/author_attestation","sign_citation":"https://pith.science/pith/EZMF6WORHUDHROLVUD7M4ITYVG/action/citation_signature","submit_replication":"https://pith.science/pith/EZMF6WORHUDHROLVUD7M4ITYVG/action/replication_record"}},"created_at":"2026-07-05T08:45:29.301033+00:00","updated_at":"2026-07-05T08:45:29.301033+00:00"}