{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:YW6NOJQWZ24O6LTAGJX4RTJGK5","short_pith_number":"pith:YW6NOJQW","schema_version":"1.0","canonical_sha256":"c5bcd72616ceb8ef2e60326fc8cd26576c5e3d61b08c71f7137c6f646edae8f5","source":{"kind":"arxiv","id":"2503.18278","version":2},"attestation_state":"computed","paper":{"title":"TopV: Compatible Token Pruning with Inference Time Optimization for Fast and Low-Memory Multimodal Vision Language Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Bo Yuan, Chendi Li, Cheng Yang, Jinghua Yan, Jinqi Xiao, Lingyi Huang, Ponnuswamy Sadayappan, Xia Hu, Yang Sui, Yu Bai, Yu Gong","submitted_at":"2025-03-24T01:47:26Z","abstract_excerpt":"Vision-Language Models (VLMs) demand substantial computational resources during inference, largely due to the extensive visual input tokens for representing visual information. Previous studies have noted that visual tokens tend to receive less attention than text tokens, suggesting their lower importance during inference and potential for pruning. However, their methods encounter several challenges: reliance on greedy heuristic criteria for token importance and incompatibility with FlashAttention and KV cache. To address these issues, we introduce \\textbf{TopV}, a compatible \\textbf{TO}ken \\t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.18278","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-03-24T01:47:26Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"e6c3fee5a555b8396515f7be73d6d89bd24a80f1e25644083c44850af99db93f","abstract_canon_sha256":"6b856df3c6e67d1ecd7559dc0ab1e5ac34e86ec288abeb62766d74068a9abb2e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:41:22.452341Z","signature_b64":"UpsxqNCcAcpw8cbDgQrj6yaE8PjeC3/bbt/iKYE16dYHRZSNxWTycqeZPMBKtno2IGvMLnGfutx3MfORo43SBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c5bcd72616ceb8ef2e60326fc8cd26576c5e3d61b08c71f7137c6f646edae8f5","last_reissued_at":"2026-07-05T10:41:22.451799Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:41:22.451799Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TopV: Compatible Token Pruning with Inference Time Optimization for Fast and Low-Memory Multimodal Vision Language Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Bo Yuan, Chendi Li, Cheng Yang, Jinghua Yan, Jinqi Xiao, Lingyi Huang, Ponnuswamy Sadayappan, Xia Hu, Yang Sui, Yu Bai, Yu Gong","submitted_at":"2025-03-24T01:47:26Z","abstract_excerpt":"Vision-Language Models (VLMs) demand substantial computational resources during inference, largely due to the extensive visual input tokens for representing visual information. Previous studies have noted that visual tokens tend to receive less attention than text tokens, suggesting their lower importance during inference and potential for pruning. However, their methods encounter several challenges: reliance on greedy heuristic criteria for token importance and incompatibility with FlashAttention and KV cache. To address these issues, we introduce \\textbf{TopV}, a compatible \\textbf{TO}ken \\t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.18278","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.18278/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.18278","created_at":"2026-07-05T10:41:22.451864+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.18278v2","created_at":"2026-07-05T10:41:22.451864+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.18278","created_at":"2026-07-05T10:41:22.451864+00:00"},{"alias_kind":"pith_short_12","alias_value":"YW6NOJQWZ24O","created_at":"2026-07-05T10:41:22.451864+00:00"},{"alias_kind":"pith_short_16","alias_value":"YW6NOJQWZ24O6LTA","created_at":"2026-07-05T10:41:22.451864+00:00"},{"alias_kind":"pith_short_8","alias_value":"YW6NOJQW","created_at":"2026-07-05T10:41:22.451864+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.18788","citing_title":"Efficient Mixture-of-Experts LLM Inference with Apple Silicon NPUs","ref_index":52,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YW6NOJQWZ24O6LTAGJX4RTJGK5","json":"https://pith.science/pith/YW6NOJQWZ24O6LTAGJX4RTJGK5.json","graph_json":"https://pith.science/api/pith-number/YW6NOJQWZ24O6LTAGJX4RTJGK5/graph.json","events_json":"https://pith.science/api/pith-number/YW6NOJQWZ24O6LTAGJX4RTJGK5/events.json","paper":"https://pith.science/paper/YW6NOJQW"},"agent_actions":{"view_html":"https://pith.science/pith/YW6NOJQWZ24O6LTAGJX4RTJGK5","download_json":"https://pith.science/pith/YW6NOJQWZ24O6LTAGJX4RTJGK5.json","view_paper":"https://pith.science/paper/YW6NOJQW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.18278&json=true","fetch_graph":"https://pith.science/api/pith-number/YW6NOJQWZ24O6LTAGJX4RTJGK5/graph.json","fetch_events":"https://pith.science/api/pith-number/YW6NOJQWZ24O6LTAGJX4RTJGK5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YW6NOJQWZ24O6LTAGJX4RTJGK5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YW6NOJQWZ24O6LTAGJX4RTJGK5/action/storage_attestation","attest_author":"https://pith.science/pith/YW6NOJQWZ24O6LTAGJX4RTJGK5/action/author_attestation","sign_citation":"https://pith.science/pith/YW6NOJQWZ24O6LTAGJX4RTJGK5/action/citation_signature","submit_replication":"https://pith.science/pith/YW6NOJQWZ24O6LTAGJX4RTJGK5/action/replication_record"}},"created_at":"2026-07-05T10:41:22.451864+00:00","updated_at":"2026-07-05T10:41:22.451864+00:00"}