{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:S3677HHVBC7VGES5M4AWTKHF5I","short_pith_number":"pith:S3677HHV","schema_version":"1.0","canonical_sha256":"96fdff9cf508bf53125d670169a8e5ea220379048453496b20a62f6afda79dfd","source":{"kind":"arxiv","id":"2403.05527","version":4},"attestation_state":"computed","paper":{"title":"GEAR: An Efficient KV Cache Compression Recipe for Near-Lossless Generative Inference of LLM","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Geonhwa Jeong, Hao Kang, Qingru Zhang, Souvik Kundu, Tuo Zhao, Tushar Krishna, Zaoxing Liu","submitted_at":"2024-03-08T18:48:30Z","abstract_excerpt":"Key-value (KV) caching has become the de-facto to accelerate generation speed for large language models (LLMs) inference. However, the growing cache demand with increasing sequence length has transformed LLM inference to be a memory bound problem, significantly constraining the system throughput. Existing methods rely on dropping unimportant tokens or quantizing all entries uniformly. Such methods, however, often incur high approximation errors to represent the compressed matrices. The autoregressive decoding process further compounds the error of each step, resulting in critical deviation in "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.05527","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-03-08T18:48:30Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"dfbc3a2b90a3d8278304d09430d0ead5c7417a17b9e3591857f5d164865003ea","abstract_canon_sha256":"9cd3d644f3fc6311b5cbbee49e5f045c564394592c285c15adcf0d6fdc2b7490"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:13:50.624883Z","signature_b64":"PTbYQnb95OJxFrS87zQ8vdpbUOdngmndwBa2OPKZA5PJ0xEmZXY4ggeeMiI+EYxB6b/rRpgB4L/oMm/+DskJBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"96fdff9cf508bf53125d670169a8e5ea220379048453496b20a62f6afda79dfd","last_reissued_at":"2026-07-05T09:13:50.624351Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:13:50.624351Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GEAR: An Efficient KV Cache Compression Recipe for Near-Lossless Generative Inference of LLM","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Geonhwa Jeong, Hao Kang, Qingru Zhang, Souvik Kundu, Tuo Zhao, Tushar Krishna, Zaoxing Liu","submitted_at":"2024-03-08T18:48:30Z","abstract_excerpt":"Key-value (KV) caching has become the de-facto to accelerate generation speed for large language models (LLMs) inference. However, the growing cache demand with increasing sequence length has transformed LLM inference to be a memory bound problem, significantly constraining the system throughput. Existing methods rely on dropping unimportant tokens or quantizing all entries uniformly. Such methods, however, often incur high approximation errors to represent the compressed matrices. The autoregressive decoding process further compounds the error of each step, resulting in critical deviation in "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.05527","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.05527/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.05527","created_at":"2026-07-05T09:13:50.624413+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.05527v4","created_at":"2026-07-05T09:13:50.624413+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.05527","created_at":"2026-07-05T09:13:50.624413+00:00"},{"alias_kind":"pith_short_12","alias_value":"S3677HHVBC7V","created_at":"2026-07-05T09:13:50.624413+00:00"},{"alias_kind":"pith_short_16","alias_value":"S3677HHVBC7VGES5","created_at":"2026-07-05T09:13:50.624413+00:00"},{"alias_kind":"pith_short_8","alias_value":"S3677HHV","created_at":"2026-07-05T09:13:50.624413+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":29,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2607.08032","citing_title":"What to Keep, What to Forget: A Rate--Distortion View of Memory Compaction in LLMs and Agents","ref_index":56,"is_internal_anchor":true},{"citing_arxiv_id":"2607.08057","citing_title":"Towards Efficient Large Language Model Serving: A Survey on System-Aware KV Cache Optimization","ref_index":5,"is_internal_anchor":true},{"citing_arxiv_id":"2606.24467","citing_title":"CompressKV: Semantic-Retrieval-Guided KV-Cache Compression for Resource-Efficient Long-Context LLM Inference","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.24033","citing_title":"RoPE-Aware Bit Allocation for KV-Cache Quantization","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2606.23581","citing_title":"Kamera: Unified Position-Invariant Multimodal KV Cache for Training-Free Reuse","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08302","citing_title":"HACK++: Towards More Effective Head-Aware Key-Value Compression for Efficient Visual Autoregressive Modeling","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09916","citing_title":"IntentKV: Cross-Turn Intent-Aware KV Cache Pruning for Agent Inference","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00760","citing_title":"MosaicKV: Serving Long-Context LLM with Dynamic Two-D KV Cache Compression","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04557","citing_title":"Cartridges at Scale: Training Modular KV Caches over Large Document Collections","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18856","citing_title":"SPHERICAL KV: Angle-Domain Attention and Rate-Distortion Retention for Efficient Long-Context Inference","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00365","citing_title":"SPARQLe: Sub-Precision Activation Representation for Quantized LLM Inference","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23258","citing_title":"A Simple Plug-in for Improving Eviction-Based KV Cache Compression","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2503.19950","citing_title":"LogQuant: Log-Distributed 2-Bit Quantization of KV Cache with Superior Accuracy Preservation","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2508.15601","citing_title":"LMDeploy Accelerates Mixed-Precision LLM Inference with TurboMind","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18856","citing_title":"SPHERICAL KV: Angle-Domain Attention and Rate-Distortion Retention for Efficient Long-Context Inference","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17757","citing_title":"OSCAR: Offline Spectral Covariance-Aware Rotation for 2-bit KV Cache Quantization","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2504.19874","citing_title":"TurboQuant: Online Vector Quantization with Near-optimal Distortion Rate","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2602.02958","citing_title":"Quant VideoGen: Auto-Regressive Long Video Generation via 2-Bit KV-Cache Quantization","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02525","citing_title":"AdaHOP: Fast and Accurate Low-Precision Training via Outlier-Pattern-Aware Rotation","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2403.14608","citing_title":"Parameter-Efficient Fine-Tuning for Large Models: A Comprehensive Survey","ref_index":183,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12464","citing_title":"Search Your Block Floating Point Scales!","ref_index":129,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11478","citing_title":"FibQuant: Universal Vector Quantization for Random-Access KV-Cache Compression","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03562","citing_title":"HeadQ: Model-Visible Distortion and Score-Space Correction for KV-Cache Quantization","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2312.07104","citing_title":"SGLang: Efficient Execution of Structured Language Model Programs","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11501","citing_title":"Quantization Dominates Rank Reduction for KV-Cache Compression","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/S3677HHVBC7VGES5M4AWTKHF5I","json":"https://pith.science/pith/S3677HHVBC7VGES5M4AWTKHF5I.json","graph_json":"https://pith.science/api/pith-number/S3677HHVBC7VGES5M4AWTKHF5I/graph.json","events_json":"https://pith.science/api/pith-number/S3677HHVBC7VGES5M4AWTKHF5I/events.json","paper":"https://pith.science/paper/S3677HHV"},"agent_actions":{"view_html":"https://pith.science/pith/S3677HHVBC7VGES5M4AWTKHF5I","download_json":"https://pith.science/pith/S3677HHVBC7VGES5M4AWTKHF5I.json","view_paper":"https://pith.science/paper/S3677HHV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.05527&json=true","fetch_graph":"https://pith.science/api/pith-number/S3677HHVBC7VGES5M4AWTKHF5I/graph.json","fetch_events":"https://pith.science/api/pith-number/S3677HHVBC7VGES5M4AWTKHF5I/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/S3677HHVBC7VGES5M4AWTKHF5I/action/timestamp_anchor","attest_storage":"https://pith.science/pith/S3677HHVBC7VGES5M4AWTKHF5I/action/storage_attestation","attest_author":"https://pith.science/pith/S3677HHVBC7VGES5M4AWTKHF5I/action/author_attestation","sign_citation":"https://pith.science/pith/S3677HHVBC7VGES5M4AWTKHF5I/action/citation_signature","submit_replication":"https://pith.science/pith/S3677HHVBC7VGES5M4AWTKHF5I/action/replication_record"}},"created_at":"2026-07-05T09:13:50.624413+00:00","updated_at":"2026-07-05T09:13:50.624413+00:00"}