{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:W5XK3MCJCYJVWIB7LZWWBOMYCP","short_pith_number":"pith:W5XK3MCJ","schema_version":"1.0","canonical_sha256":"b76eadb04916135b203f5e6d60b99813f4a20dd99f0c49e9f00344b285392ce8","source":{"kind":"arxiv","id":"2411.10803","version":1},"attestation_state":"computed","paper":{"title":"Multi-Stage Vision Token Dropping: Towards Efficient Multimodal Large Language Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Liangtao Shi, Linfeng Zhang, Quanjun Yin, Richang Hong, Ting Liu, Yue Hu","submitted_at":"2024-11-16T13:45:33Z","abstract_excerpt":"The vision tokens in multimodal large language models usually exhibit significant spatial and temporal redundancy and take up most of the input tokens, which harms their inference efficiency. To solve this problem, some recent works were introduced to drop the unimportant tokens during inference where the importance of each token is decided only by the information in either the vision encoding stage or the prefilling stage. In this paper, we propose Multi-stage Token Dropping (MustDrop) to measure the importance of each token from the whole lifecycle, including the vision encoding stage, prefi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.10803","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-11-16T13:45:33Z","cross_cats_sorted":[],"title_canon_sha256":"6e9c46b4cbbc3f3db2a0a60027f4fe61967ba2aa96114a47049c338383f87975","abstract_canon_sha256":"3b583b2b2d9c1d7d6c2fc63815c23a3aaaf6c2ecc22b967048667168c7ac52c7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:36:22.403539Z","signature_b64":"Ut2RiEIKXJ7TUvr3qYGbmjNSuJRKe+5xoSpqVU46ZJjsOjZVbamsKZ8KiDsaDwShETZ2H+GUUprbUGIYymy4Dw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b76eadb04916135b203f5e6d60b99813f4a20dd99f0c49e9f00344b285392ce8","last_reissued_at":"2026-07-05T09:36:22.403019Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:36:22.403019Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Multi-Stage Vision Token Dropping: Towards Efficient Multimodal Large Language Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Liangtao Shi, Linfeng Zhang, Quanjun Yin, Richang Hong, Ting Liu, Yue Hu","submitted_at":"2024-11-16T13:45:33Z","abstract_excerpt":"The vision tokens in multimodal large language models usually exhibit significant spatial and temporal redundancy and take up most of the input tokens, which harms their inference efficiency. To solve this problem, some recent works were introduced to drop the unimportant tokens during inference where the importance of each token is decided only by the information in either the vision encoding stage or the prefilling stage. In this paper, we propose Multi-stage Token Dropping (MustDrop) to measure the importance of each token from the whole lifecycle, including the vision encoding stage, prefi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.10803","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.10803/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.10803","created_at":"2026-07-05T09:36:22.403080+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.10803v1","created_at":"2026-07-05T09:36:22.403080+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.10803","created_at":"2026-07-05T09:36:22.403080+00:00"},{"alias_kind":"pith_short_12","alias_value":"W5XK3MCJCYJV","created_at":"2026-07-05T09:36:22.403080+00:00"},{"alias_kind":"pith_short_16","alias_value":"W5XK3MCJCYJVWIB7","created_at":"2026-07-05T09:36:22.403080+00:00"},{"alias_kind":"pith_short_8","alias_value":"W5XK3MCJ","created_at":"2026-07-05T09:36:22.403080+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.27161","citing_title":"TOPS: First-Principles Visual Token Pruning via Constructing Token Optimal Preservation Sets for Efficient MLLM Inference","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12412","citing_title":"Reroute, Don't Remove: Recoverable Visual Token Routing for Vision-Language Models","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31383","citing_title":"MS-Resampler: Multi-Scope Visual Resampling for Efficient Multimodal LLMs","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30126","citing_title":"PARCEL: Pool-Anchored Resampling with Conditioned Elastic Queries for Efficient Vision-Language Understanding","ref_index":74,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30010","citing_title":"EarlyTom: Early Token Compression Completes Fast Video Understanding","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2503.14075","citing_title":"Growing a Multi-head Twig via Distillation and Reinforcement Learning to Accelerate Large Vision-Language Models","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2601.21531","citing_title":"On the Adversarial Robustness of Large Vision-Language Models under Visual Token Compression","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17447","citing_title":"FastOCR: Dynamic Visual Fixation via KV Cache Pruning for Efficient Document Parsing","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19218","citing_title":"Rotation-Aligned Key Channel Pruning for Efficient Vision-Language Model Inference","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2603.01400","citing_title":"Token Reduction via Local and Global Contexts Optimization for Efficient Video Large Language Models","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11627","citing_title":"POINTS-Long: Adaptive Dual-Mode Visual Reasoning in MLLMs","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17087","citing_title":"EvoComp: Learning Visual Token Compression for Multimodal Large Language Models via Semantic-Guided Evolutionary Labeling","ref_index":33,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/W5XK3MCJCYJVWIB7LZWWBOMYCP","json":"https://pith.science/pith/W5XK3MCJCYJVWIB7LZWWBOMYCP.json","graph_json":"https://pith.science/api/pith-number/W5XK3MCJCYJVWIB7LZWWBOMYCP/graph.json","events_json":"https://pith.science/api/pith-number/W5XK3MCJCYJVWIB7LZWWBOMYCP/events.json","paper":"https://pith.science/paper/W5XK3MCJ"},"agent_actions":{"view_html":"https://pith.science/pith/W5XK3MCJCYJVWIB7LZWWBOMYCP","download_json":"https://pith.science/pith/W5XK3MCJCYJVWIB7LZWWBOMYCP.json","view_paper":"https://pith.science/paper/W5XK3MCJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.10803&json=true","fetch_graph":"https://pith.science/api/pith-number/W5XK3MCJCYJVWIB7LZWWBOMYCP/graph.json","fetch_events":"https://pith.science/api/pith-number/W5XK3MCJCYJVWIB7LZWWBOMYCP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/W5XK3MCJCYJVWIB7LZWWBOMYCP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/W5XK3MCJCYJVWIB7LZWWBOMYCP/action/storage_attestation","attest_author":"https://pith.science/pith/W5XK3MCJCYJVWIB7LZWWBOMYCP/action/author_attestation","sign_citation":"https://pith.science/pith/W5XK3MCJCYJVWIB7LZWWBOMYCP/action/citation_signature","submit_replication":"https://pith.science/pith/W5XK3MCJCYJVWIB7LZWWBOMYCP/action/replication_record"}},"created_at":"2026-07-05T09:36:22.403080+00:00","updated_at":"2026-07-05T09:36:22.403080+00:00"}