{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:OI4UPZPS744ZEDC4WVBF5YXC7N","short_pith_number":"pith:OI4UPZPS","schema_version":"1.0","canonical_sha256":"723947e5f2ff39920c5cb5425ee2e2fb45a9b4827c0a67a943677fae6c2331d9","source":{"kind":"arxiv","id":"2509.00419","version":1},"attestation_state":"computed","paper":{"title":"LightVLM: Acceleraing Large Multimodal Models with Pyramid Token Merging and KV Cache Compression","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Fanhua Shang, Liang Wan, Lianyu Hu, Wei Feng","submitted_at":"2025-08-30T08:57:53Z","abstract_excerpt":"In this paper, we introduce LightVLM, a simple but effective method that can be seamlessly deployed upon existing Vision-Language Models (VLMs) to greatly accelerate the inference process in a training-free manner. We divide the inference procedure of VLMs into two stages, i.e., encoding and decoding, and propose to simultaneously accelerate VLMs in both stages to largely improve model efficiency. During encoding, we propose pyramid token merging to reduce tokens of different LLM layers in a hierarchical manner by finally only keeping a few dominant tokens to achieve high efficiency. During de"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2509.00419","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-08-30T08:57:53Z","cross_cats_sorted":[],"title_canon_sha256":"54528fa49e6cd4f4245e10e4c2c3aca91344c902cbf31f945eea9ceb4e7ebefe","abstract_canon_sha256":"df99a1ee1722600009ae85972cf39e1ee7e9084e797992fadeb85a1e9a2bc156"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:02:15.901806Z","signature_b64":"k4/qy7DiHqfKbKsiiX1MZiQUAc7GgL87VDU5+uhzDgRJRyN4DXi84mI+wQgl6/I9U3hcHBkmS7vyDKS9u7A8CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"723947e5f2ff39920c5cb5425ee2e2fb45a9b4827c0a67a943677fae6c2331d9","last_reissued_at":"2026-07-05T12:02:15.901276Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:02:15.901276Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LightVLM: Acceleraing Large Multimodal Models with Pyramid Token Merging and KV Cache Compression","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Fanhua Shang, Liang Wan, Lianyu Hu, Wei Feng","submitted_at":"2025-08-30T08:57:53Z","abstract_excerpt":"In this paper, we introduce LightVLM, a simple but effective method that can be seamlessly deployed upon existing Vision-Language Models (VLMs) to greatly accelerate the inference process in a training-free manner. We divide the inference procedure of VLMs into two stages, i.e., encoding and decoding, and propose to simultaneously accelerate VLMs in both stages to largely improve model efficiency. During encoding, we propose pyramid token merging to reduce tokens of different LLM layers in a hierarchical manner by finally only keeping a few dominant tokens to achieve high efficiency. During de"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.00419","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.00419/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2509.00419","created_at":"2026-07-05T12:02:15.901350+00:00"},{"alias_kind":"arxiv_version","alias_value":"2509.00419v1","created_at":"2026-07-05T12:02:15.901350+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.00419","created_at":"2026-07-05T12:02:15.901350+00:00"},{"alias_kind":"pith_short_12","alias_value":"OI4UPZPS744Z","created_at":"2026-07-05T12:02:15.901350+00:00"},{"alias_kind":"pith_short_16","alias_value":"OI4UPZPS744ZEDC4","created_at":"2026-07-05T12:02:15.901350+00:00"},{"alias_kind":"pith_short_8","alias_value":"OI4UPZPS","created_at":"2026-07-05T12:02:15.901350+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24165","citing_title":"Spectral Evolution-Guided Token Pruning in Multimodal Large Language Models","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2606.24156","citing_title":"Accelerating Multimodal Large Language Models with Prior-Corrected Token Reduction","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28115","citing_title":"CIVIC: End-to-End Sequence Compactness for Efficient Vision-Language Models","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05887","citing_title":"HybridKV: Hybrid KV Cache Compression for Efficient Multimodal Large Language Model Inference","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OI4UPZPS744ZEDC4WVBF5YXC7N","json":"https://pith.science/pith/OI4UPZPS744ZEDC4WVBF5YXC7N.json","graph_json":"https://pith.science/api/pith-number/OI4UPZPS744ZEDC4WVBF5YXC7N/graph.json","events_json":"https://pith.science/api/pith-number/OI4UPZPS744ZEDC4WVBF5YXC7N/events.json","paper":"https://pith.science/paper/OI4UPZPS"},"agent_actions":{"view_html":"https://pith.science/pith/OI4UPZPS744ZEDC4WVBF5YXC7N","download_json":"https://pith.science/pith/OI4UPZPS744ZEDC4WVBF5YXC7N.json","view_paper":"https://pith.science/paper/OI4UPZPS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2509.00419&json=true","fetch_graph":"https://pith.science/api/pith-number/OI4UPZPS744ZEDC4WVBF5YXC7N/graph.json","fetch_events":"https://pith.science/api/pith-number/OI4UPZPS744ZEDC4WVBF5YXC7N/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OI4UPZPS744ZEDC4WVBF5YXC7N/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OI4UPZPS744ZEDC4WVBF5YXC7N/action/storage_attestation","attest_author":"https://pith.science/pith/OI4UPZPS744ZEDC4WVBF5YXC7N/action/author_attestation","sign_citation":"https://pith.science/pith/OI4UPZPS744ZEDC4WVBF5YXC7N/action/citation_signature","submit_replication":"https://pith.science/pith/OI4UPZPS744ZEDC4WVBF5YXC7N/action/replication_record"}},"created_at":"2026-07-05T12:02:15.901350+00:00","updated_at":"2026-07-05T12:02:15.901350+00:00"}