{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:2I6XLRR7FDHUCCRRHTMUTVZFM5","short_pith_number":"pith:2I6XLRR7","schema_version":"1.0","canonical_sha256":"d23d75c63f28cf410a313cd949d725674e3630a504a6869898aae28c8fb1f282","source":{"kind":"arxiv","id":"2409.10197","version":2},"attestation_state":"computed","paper":{"title":"Fit and Prune: Fast and Training-free Visual Token Pruning for Multi-modal Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.MM"],"primary_cat":"cs.CV","authors_text":"Qiong Wu, Weihao Ye, Wenhao Lin, Yiyi Zhou","submitted_at":"2024-09-16T11:43:19Z","abstract_excerpt":"Recent progress in Multimodal Large Language Models(MLLMs) often use large image tokens to compensate the visual shortcoming of MLLMs, which not only exhibits obvious redundancy but also greatly exacerbates the already high computation. Token pruning is an effective solution for speeding up MLLMs, but when and how to drop tokens still remains a challenge. In this paper, we propose a novel and training-free approach for the effective visual token pruning of MLLMs, termed FitPrune, which can quickly produce a complete pruning recipe for MLLMs according to a pre-defined budget. Specifically, FitP"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.10197","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-09-16T11:43:19Z","cross_cats_sorted":["cs.CL","cs.MM"],"title_canon_sha256":"76c06a23dea201701ecd79fa6243588c6dc4afbe97d41730d4d8cac64a2b1cb6","abstract_canon_sha256":"cbbb3206683e5c03f60f8c0860e9dec0e0160d536ca3e6b5f57af31940ba3f95"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:54:04.435554Z","signature_b64":"1h4nF4g6Ihjt1fw+PAqSCu6O0T85N32Ye4NAnJYi8xlqI1xmix2Qzq39Z0BobdQzjpm5RX/jadXpHmRxeDQQAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d23d75c63f28cf410a313cd949d725674e3630a504a6869898aae28c8fb1f282","last_reissued_at":"2026-07-05T09:54:04.435098Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:54:04.435098Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Fit and Prune: Fast and Training-free Visual Token Pruning for Multi-modal Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.MM"],"primary_cat":"cs.CV","authors_text":"Qiong Wu, Weihao Ye, Wenhao Lin, Yiyi Zhou","submitted_at":"2024-09-16T11:43:19Z","abstract_excerpt":"Recent progress in Multimodal Large Language Models(MLLMs) often use large image tokens to compensate the visual shortcoming of MLLMs, which not only exhibits obvious redundancy but also greatly exacerbates the already high computation. Token pruning is an effective solution for speeding up MLLMs, but when and how to drop tokens still remains a challenge. In this paper, we propose a novel and training-free approach for the effective visual token pruning of MLLMs, termed FitPrune, which can quickly produce a complete pruning recipe for MLLMs according to a pre-defined budget. Specifically, FitP"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.10197","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.10197/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.10197","created_at":"2026-07-05T09:54:04.435156+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.10197v2","created_at":"2026-07-05T09:54:04.435156+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.10197","created_at":"2026-07-05T09:54:04.435156+00:00"},{"alias_kind":"pith_short_12","alias_value":"2I6XLRR7FDHU","created_at":"2026-07-05T09:54:04.435156+00:00"},{"alias_kind":"pith_short_16","alias_value":"2I6XLRR7FDHUCCRR","created_at":"2026-07-05T09:54:04.435156+00:00"},{"alias_kind":"pith_short_8","alias_value":"2I6XLRR7","created_at":"2026-07-05T09:54:04.435156+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.31383","citing_title":"MS-Resampler: Multi-Scope Visual Resampling for Efficient Multimodal LLMs","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25343","citing_title":"Toward Native Multimodal Modeling: A Roadmap","ref_index":204,"is_internal_anchor":false},{"citing_arxiv_id":"2503.14075","citing_title":"Growing a Multi-head Twig via Distillation and Reinforcement Learning to Accelerate Large Vision-Language Models","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2603.22911","citing_title":"ForestPrune: High-ratio Visual Token Compression for Video Multimodal Large Language Models via Spatial-Temporal Forest Modeling","ref_index":61,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2I6XLRR7FDHUCCRRHTMUTVZFM5","json":"https://pith.science/pith/2I6XLRR7FDHUCCRRHTMUTVZFM5.json","graph_json":"https://pith.science/api/pith-number/2I6XLRR7FDHUCCRRHTMUTVZFM5/graph.json","events_json":"https://pith.science/api/pith-number/2I6XLRR7FDHUCCRRHTMUTVZFM5/events.json","paper":"https://pith.science/paper/2I6XLRR7"},"agent_actions":{"view_html":"https://pith.science/pith/2I6XLRR7FDHUCCRRHTMUTVZFM5","download_json":"https://pith.science/pith/2I6XLRR7FDHUCCRRHTMUTVZFM5.json","view_paper":"https://pith.science/paper/2I6XLRR7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.10197&json=true","fetch_graph":"https://pith.science/api/pith-number/2I6XLRR7FDHUCCRRHTMUTVZFM5/graph.json","fetch_events":"https://pith.science/api/pith-number/2I6XLRR7FDHUCCRRHTMUTVZFM5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2I6XLRR7FDHUCCRRHTMUTVZFM5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2I6XLRR7FDHUCCRRHTMUTVZFM5/action/storage_attestation","attest_author":"https://pith.science/pith/2I6XLRR7FDHUCCRRHTMUTVZFM5/action/author_attestation","sign_citation":"https://pith.science/pith/2I6XLRR7FDHUCCRRHTMUTVZFM5/action/citation_signature","submit_replication":"https://pith.science/pith/2I6XLRR7FDHUCCRRHTMUTVZFM5/action/replication_record"}},"created_at":"2026-07-05T09:54:04.435156+00:00","updated_at":"2026-07-05T09:54:04.435156+00:00"}