{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:LV63QY2TKBSOIDCVTAQSFH5YHE","short_pith_number":"pith:LV63QY2T","schema_version":"1.0","canonical_sha256":"5d7db863535064e40c559821229fb8392ba401aa7e3e879149b776fa924f7a82","source":{"kind":"arxiv","id":"2501.15021","version":1},"attestation_state":"computed","paper":{"title":"AKVQ-VL: Attention-Aware KV Cache Adaptive 2-Bit Quantization for Vision-Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hanyu Wei, Huangqi Yu, Kehong Yuan, Linge Li, Wang Shen, Zhe Chen, Zunhai Su","submitted_at":"2025-01-25T02:01:56Z","abstract_excerpt":"Vision-language models (VLMs) show remarkable performance in multimodal tasks. However, excessively long multimodal inputs lead to oversized Key-Value (KV) caches, resulting in significant memory consumption and I/O bottlenecks. Previous KV quantization methods for Large Language Models (LLMs) may alleviate these issues but overlook the attention saliency differences of multimodal tokens, resulting in suboptimal performance. In this paper, we investigate the attention-aware token saliency patterns in VLM and propose AKVQ-VL. AKVQ-VL leverages the proposed Text-Salient Attention (TSA) and Pivot"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.15021","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-01-25T02:01:56Z","cross_cats_sorted":[],"title_canon_sha256":"d570ee20808418c1fb1d8e999b59fc63ed1d719cabad326e3b14575d1dbd2af5","abstract_canon_sha256":"8e3f14384e93ddc581b81c45aaa52b671791974f776e06e77aa3af85f9117162"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:05:22.330911Z","signature_b64":"NM9KMufrEeL216p1M1+Z3ulDqJFokit/nfCVGlapzGyGSz5e+kNcKV//slO2xZA9AYIma/WnI9Te8nobr/KgDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5d7db863535064e40c559821229fb8392ba401aa7e3e879149b776fa924f7a82","last_reissued_at":"2026-07-05T10:05:22.330502Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:05:22.330502Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AKVQ-VL: Attention-Aware KV Cache Adaptive 2-Bit Quantization for Vision-Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hanyu Wei, Huangqi Yu, Kehong Yuan, Linge Li, Wang Shen, Zhe Chen, Zunhai Su","submitted_at":"2025-01-25T02:01:56Z","abstract_excerpt":"Vision-language models (VLMs) show remarkable performance in multimodal tasks. However, excessively long multimodal inputs lead to oversized Key-Value (KV) caches, resulting in significant memory consumption and I/O bottlenecks. Previous KV quantization methods for Large Language Models (LLMs) may alleviate these issues but overlook the attention saliency differences of multimodal tokens, resulting in suboptimal performance. In this paper, we investigate the attention-aware token saliency patterns in VLM and propose AKVQ-VL. AKVQ-VL leverages the proposed Text-Salient Attention (TSA) and Pivot"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.15021","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.15021/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.15021","created_at":"2026-07-05T10:05:22.330560+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.15021v1","created_at":"2026-07-05T10:05:22.330560+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.15021","created_at":"2026-07-05T10:05:22.330560+00:00"},{"alias_kind":"pith_short_12","alias_value":"LV63QY2TKBSO","created_at":"2026-07-05T10:05:22.330560+00:00"},{"alias_kind":"pith_short_16","alias_value":"LV63QY2TKBSOIDCV","created_at":"2026-07-05T10:05:22.330560+00:00"},{"alias_kind":"pith_short_8","alias_value":"LV63QY2T","created_at":"2026-07-05T10:05:22.330560+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.16439","citing_title":"KVCapsule: Efficient Sequential KV Cache Compression for Vision-Language Models with Asymmetric Redundancy","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2602.10718","citing_title":"SnapMLA: Efficient Long-Context MLA Decoding via Hardware-Aware FP8 Quantized Pipelining","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02262","citing_title":"WindowQuant: Mixed-Precision KV Cache Quantization based on Window-Level Similarity for VLMs Inference Optimization","ref_index":40,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LV63QY2TKBSOIDCVTAQSFH5YHE","json":"https://pith.science/pith/LV63QY2TKBSOIDCVTAQSFH5YHE.json","graph_json":"https://pith.science/api/pith-number/LV63QY2TKBSOIDCVTAQSFH5YHE/graph.json","events_json":"https://pith.science/api/pith-number/LV63QY2TKBSOIDCVTAQSFH5YHE/events.json","paper":"https://pith.science/paper/LV63QY2T"},"agent_actions":{"view_html":"https://pith.science/pith/LV63QY2TKBSOIDCVTAQSFH5YHE","download_json":"https://pith.science/pith/LV63QY2TKBSOIDCVTAQSFH5YHE.json","view_paper":"https://pith.science/paper/LV63QY2T","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.15021&json=true","fetch_graph":"https://pith.science/api/pith-number/LV63QY2TKBSOIDCVTAQSFH5YHE/graph.json","fetch_events":"https://pith.science/api/pith-number/LV63QY2TKBSOIDCVTAQSFH5YHE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LV63QY2TKBSOIDCVTAQSFH5YHE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LV63QY2TKBSOIDCVTAQSFH5YHE/action/storage_attestation","attest_author":"https://pith.science/pith/LV63QY2TKBSOIDCVTAQSFH5YHE/action/author_attestation","sign_citation":"https://pith.science/pith/LV63QY2TKBSOIDCVTAQSFH5YHE/action/citation_signature","submit_replication":"https://pith.science/pith/LV63QY2TKBSOIDCVTAQSFH5YHE/action/replication_record"}},"created_at":"2026-07-05T10:05:22.330560+00:00","updated_at":"2026-07-05T10:05:22.330560+00:00"}