{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:72FVA3SVNWAVZTRSSIXQYBN2HK","short_pith_number":"pith:72FVA3SV","schema_version":"1.0","canonical_sha256":"fe8b506e556d815cce32922f0c05ba3ab91b40ccb65ab9387d39044ec6f5f66d","source":{"kind":"arxiv","id":"2502.02617","version":1},"attestation_state":"computed","paper":{"title":"PolarQuant: Quantizing KV Caches with Polar Transformation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Amin Karbasi, Amir Zandieh, Insu Han, Praneeth Kacham, Vahab Mirrokni","submitted_at":"2025-02-04T08:52:13Z","abstract_excerpt":"Large language models (LLMs) require significant memory to store Key-Value (KV) embeddings in their KV cache, especially when handling long-range contexts. Quantization of these KV embeddings is a common technique to reduce memory consumption. This work introduces PolarQuant, a novel quantization method employing random preconditioning and polar transformation. Our method transforms the KV embeddings into polar coordinates using an efficient recursive algorithm and then quantizes resulting angles. Our key insight is that, after random preconditioning, the angles in the polar representation exh"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.02617","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-04T08:52:13Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"69f7dde79f63bcd716c24b573e5dde29e9d18513c757241ea96a7e748218a311","abstract_canon_sha256":"798cc44b3deb17c7cbc898162b67cd46f21ac095339edc87caddfe76a20759a4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:09:31.740392Z","signature_b64":"CTnaevo5QdAP6NRaUIyPE1NxzXVtF48zQDRUMTe8lXftC6vc0pmtXRwJUCo7d3Sh/eC0PnxxYXBKc7VTs2DMCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fe8b506e556d815cce32922f0c05ba3ab91b40ccb65ab9387d39044ec6f5f66d","last_reissued_at":"2026-07-05T10:09:31.739932Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:09:31.739932Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"PolarQuant: Quantizing KV Caches with Polar Transformation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Amin Karbasi, Amir Zandieh, Insu Han, Praneeth Kacham, Vahab Mirrokni","submitted_at":"2025-02-04T08:52:13Z","abstract_excerpt":"Large language models (LLMs) require significant memory to store Key-Value (KV) embeddings in their KV cache, especially when handling long-range contexts. Quantization of these KV embeddings is a common technique to reduce memory consumption. This work introduces PolarQuant, a novel quantization method employing random preconditioning and polar transformation. Our method transforms the KV embeddings into polar coordinates using an efficient recursive algorithm and then quantizes resulting angles. Our key insight is that, after random preconditioning, the angles in the polar representation exh"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.02617","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.02617/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.02617","created_at":"2026-07-05T10:09:31.739990+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.02617v1","created_at":"2026-07-05T10:09:31.739990+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.02617","created_at":"2026-07-05T10:09:31.739990+00:00"},{"alias_kind":"pith_short_12","alias_value":"72FVA3SVNWAV","created_at":"2026-07-05T10:09:31.739990+00:00"},{"alias_kind":"pith_short_16","alias_value":"72FVA3SVNWAVZTRS","created_at":"2026-07-05T10:09:31.739990+00:00"},{"alias_kind":"pith_short_8","alias_value":"72FVA3SV","created_at":"2026-07-05T10:09:31.739990+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.17415","citing_title":"IVF-TQ: Calibration-Free Streaming Vector Search via a Codebook-Free Residual Layer","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18856","citing_title":"SPHERICAL KV: Angle-Domain Attention and Rate-Distortion Retention for Efficient Long-Context Inference","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17415","citing_title":"IVF-TQ: Calibration-Free Streaming Vector Search via a Codebook-Free Residual Layer","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2504.19874","citing_title":"TurboQuant: Online Vector Quantization with Near-optimal Distortion Rate","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19660","citing_title":"OScaR: The Occam's Razor for Extreme KV Cache Quantization in LLMs and Beyond","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02638","citing_title":"AXELRAM: Quantize Once, Never Dequantize","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07662","citing_title":"Direction-Preserving Number Representations","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04514","citing_title":"SuperLocalMemory V3.3: The Living Brain -- Biologically-Inspired Forgetting, Cognitive Quantization, and Multi-Channel Retrieval for Zero-LLM Agent Memory Systems","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05366","citing_title":"3DTurboQuant: Training-Free Near-Optimal Quantization for 3D Reconstruction Models","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02905","citing_title":"eOptShrinkQ: Near-Lossless KV Cache Compression Through Optimal Spectral Denoising and Quantization","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14442","citing_title":"Hierarchical vs. Flat Iteration in Shared-Weight Transformers","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16957","citing_title":"Open-TQ-Metal: Fused Compressed-Domain Attention for Long-Context LLM Inference on Apple Silicon","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/72FVA3SVNWAVZTRSSIXQYBN2HK","json":"https://pith.science/pith/72FVA3SVNWAVZTRSSIXQYBN2HK.json","graph_json":"https://pith.science/api/pith-number/72FVA3SVNWAVZTRSSIXQYBN2HK/graph.json","events_json":"https://pith.science/api/pith-number/72FVA3SVNWAVZTRSSIXQYBN2HK/events.json","paper":"https://pith.science/paper/72FVA3SV"},"agent_actions":{"view_html":"https://pith.science/pith/72FVA3SVNWAVZTRSSIXQYBN2HK","download_json":"https://pith.science/pith/72FVA3SVNWAVZTRSSIXQYBN2HK.json","view_paper":"https://pith.science/paper/72FVA3SV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.02617&json=true","fetch_graph":"https://pith.science/api/pith-number/72FVA3SVNWAVZTRSSIXQYBN2HK/graph.json","fetch_events":"https://pith.science/api/pith-number/72FVA3SVNWAVZTRSSIXQYBN2HK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/72FVA3SVNWAVZTRSSIXQYBN2HK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/72FVA3SVNWAVZTRSSIXQYBN2HK/action/storage_attestation","attest_author":"https://pith.science/pith/72FVA3SVNWAVZTRSSIXQYBN2HK/action/author_attestation","sign_citation":"https://pith.science/pith/72FVA3SVNWAVZTRSSIXQYBN2HK/action/citation_signature","submit_replication":"https://pith.science/pith/72FVA3SVNWAVZTRSSIXQYBN2HK/action/replication_record"}},"created_at":"2026-07-05T10:09:31.739990+00:00","updated_at":"2026-07-05T10:09:31.739990+00:00"}