{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:7QFYZ2PX5RJJ6JIIMHVSXYDG6P","short_pith_number":"pith:7QFYZ2PX","schema_version":"1.0","canonical_sha256":"fc0b8ce9f7ec529f250861eb2be066f3d62a13c43ea7a676e2ceb9d8b027b1c4","source":{"kind":"arxiv","id":"2406.18139","version":1},"attestation_state":"computed","paper":{"title":"LOOK-M: Look-Once Optimization in KV Cache for Efficient Multimodal Long-Context Inference","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Che Liu, Jinfa Huang, Li Yuan, Longyue Wang, Peng Jin, Zhihong Zhu, Zhongwei Wan, Ziang Wu","submitted_at":"2024-06-26T07:44:24Z","abstract_excerpt":"Long-context Multimodal Large Language Models (MLLMs) demand substantial computational resources for inference as the growth of their multimodal Key-Value (KV) cache, in response to increasing input lengths, challenges memory and time efficiency. Unlike single-modality LLMs that manage only textual contexts, the KV cache of long-context MLLMs includes representations from multiple images with temporal and spatial relationships and related textual contexts. The predominance of image tokens means traditional optimizations for LLMs' KV caches are unsuitable for multimodal long-context settings, a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.18139","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-06-26T07:44:24Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"cbf780bf9988d9ed44f978e3ff74322e37b03beacb088d9f20ea04c44b121b02","abstract_canon_sha256":"c7ed3e345a6bf28d0c57064aa22cc29e397f3d39a97082558bb56cfdaf6d2053"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:36:55.834265Z","signature_b64":"4iMr5wH7bEPo51dtNt9oTpbfAPwVYVPQk1a9jP3THqJKoUi+QV4yHCOavRGF+ng+uKNwbziaYjjw+gJxt7PECg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fc0b8ce9f7ec529f250861eb2be066f3d62a13c43ea7a676e2ceb9d8b027b1c4","last_reissued_at":"2026-07-05T08:36:55.833793Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:36:55.833793Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LOOK-M: Look-Once Optimization in KV Cache for Efficient Multimodal Long-Context Inference","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Che Liu, Jinfa Huang, Li Yuan, Longyue Wang, Peng Jin, Zhihong Zhu, Zhongwei Wan, Ziang Wu","submitted_at":"2024-06-26T07:44:24Z","abstract_excerpt":"Long-context Multimodal Large Language Models (MLLMs) demand substantial computational resources for inference as the growth of their multimodal Key-Value (KV) cache, in response to increasing input lengths, challenges memory and time efficiency. Unlike single-modality LLMs that manage only textual contexts, the KV cache of long-context MLLMs includes representations from multiple images with temporal and spatial relationships and related textual contexts. The predominance of image tokens means traditional optimizations for LLMs' KV caches are unsuitable for multimodal long-context settings, a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.18139","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.18139/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.18139","created_at":"2026-07-05T08:36:55.833851+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.18139v1","created_at":"2026-07-05T08:36:55.833851+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.18139","created_at":"2026-07-05T08:36:55.833851+00:00"},{"alias_kind":"pith_short_12","alias_value":"7QFYZ2PX5RJJ","created_at":"2026-07-05T08:36:55.833851+00:00"},{"alias_kind":"pith_short_16","alias_value":"7QFYZ2PX5RJJ6JII","created_at":"2026-07-05T08:36:55.833851+00:00"},{"alias_kind":"pith_short_8","alias_value":"7QFYZ2PX","created_at":"2026-07-05T08:36:55.833851+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08032","citing_title":"What to Keep, What to Forget: A Rate--Distortion View of Memory Compaction in LLMs and Agents","ref_index":118,"is_internal_anchor":true},{"citing_arxiv_id":"2606.08302","citing_title":"HACK++: Towards More Effective Head-Aware Key-Value Compression for Efficient Visual Autoregressive Modeling","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31457","citing_title":"VisionPulse: Dynamic Visual Sparsity for Efficient Multimodal Reasoning","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16439","citing_title":"KVCapsule: Efficient Sequential KV Cache Compression for Vision-Language Models with Asymmetric Redundancy","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2410.10781","citing_title":"When Attention Sink Emerges in Language Models: An Empirical View","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25642","citing_title":"Prefill-Time Intervention for Mitigating Hallucination in Large Vision-Language Models","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06611","citing_title":"The Structural Origin of Attention Sink: Variance Discrepancy, Super Neurons, and Dimension Disparity","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11627","citing_title":"POINTS-Long: Adaptive Dual-Mode Visual Reasoning in MLLMs","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04075","citing_title":"RetentiveKV: State-Space Memory for Uncertainty-Aware Multimodal KV Cache Eviction","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16734","citing_title":"Reducing Peak Memory Usage for Modern Multimodal Large Language Model Pipelines","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18260","citing_title":"Geometry-Guided 3D Visual Token Pruning for Video-Language Models","ref_index":27,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7QFYZ2PX5RJJ6JIIMHVSXYDG6P","json":"https://pith.science/pith/7QFYZ2PX5RJJ6JIIMHVSXYDG6P.json","graph_json":"https://pith.science/api/pith-number/7QFYZ2PX5RJJ6JIIMHVSXYDG6P/graph.json","events_json":"https://pith.science/api/pith-number/7QFYZ2PX5RJJ6JIIMHVSXYDG6P/events.json","paper":"https://pith.science/paper/7QFYZ2PX"},"agent_actions":{"view_html":"https://pith.science/pith/7QFYZ2PX5RJJ6JIIMHVSXYDG6P","download_json":"https://pith.science/pith/7QFYZ2PX5RJJ6JIIMHVSXYDG6P.json","view_paper":"https://pith.science/paper/7QFYZ2PX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.18139&json=true","fetch_graph":"https://pith.science/api/pith-number/7QFYZ2PX5RJJ6JIIMHVSXYDG6P/graph.json","fetch_events":"https://pith.science/api/pith-number/7QFYZ2PX5RJJ6JIIMHVSXYDG6P/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7QFYZ2PX5RJJ6JIIMHVSXYDG6P/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7QFYZ2PX5RJJ6JIIMHVSXYDG6P/action/storage_attestation","attest_author":"https://pith.science/pith/7QFYZ2PX5RJJ6JIIMHVSXYDG6P/action/author_attestation","sign_citation":"https://pith.science/pith/7QFYZ2PX5RJJ6JIIMHVSXYDG6P/action/citation_signature","submit_replication":"https://pith.science/pith/7QFYZ2PX5RJJ6JIIMHVSXYDG6P/action/replication_record"}},"created_at":"2026-07-05T08:36:55.833851+00:00","updated_at":"2026-07-05T08:36:55.833851+00:00"}