{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:CMCU2KGTIEBAF3PUGSAIT5KISN","short_pith_number":"pith:CMCU2KGT","schema_version":"1.0","canonical_sha256":"13054d28d3410202edf4348089f548935520bfe5f22337dbadedc536317e9ba9","source":{"kind":"arxiv","id":"2506.21873","version":1},"attestation_state":"computed","paper":{"title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Chieh-Kai Lin, Hung-Jen Chen, Min Sun, Ruei-Chi Lai, Shiang-Feng Tsai, Tzu-Chun Chien","submitted_at":"2025-06-27T03:11:22Z","abstract_excerpt":"Recent Multimodal Large Language Models (MLLMs) have demonstrated strong performance in visual grounding, establishing themselves as a general interface for various vision-language applications. This progress has driven the development of token pruning methods to mitigate the high computational costs associated with processing numerous visual tokens. However, we observe that pruning significantly weakens the model's grounding ability, leading to incorrect predictions and drastic performance degradation. In Referring Expression Comprehension (REC), for instance, pruning causes the accuracy of L"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.21873","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-06-27T03:11:22Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"c973e031b595f9ead0211c19082d68e499de16ac6394a575a684e5751aade533","abstract_canon_sha256":"84fadf62830e575ad4296b13f58158a990c25e111ae5d39bac88eff7be3a60bd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:28:03.108456Z","signature_b64":"Y6ZLsoxGJMkN/z+jK39SESs5+ENw1PbiF9SIv9ObsEg0AcG+uGNDTx8CpIWO/xdrwP1/Y6bHfFciB+sjcBtaCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"13054d28d3410202edf4348089f548935520bfe5f22337dbadedc536317e9ba9","last_reissued_at":"2026-07-05T11:28:03.107939Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:28:03.107939Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Chieh-Kai Lin, Hung-Jen Chen, Min Sun, Ruei-Chi Lai, Shiang-Feng Tsai, Tzu-Chun Chien","submitted_at":"2025-06-27T03:11:22Z","abstract_excerpt":"Recent Multimodal Large Language Models (MLLMs) have demonstrated strong performance in visual grounding, establishing themselves as a general interface for various vision-language applications. This progress has driven the development of token pruning methods to mitigate the high computational costs associated with processing numerous visual tokens. However, we observe that pruning significantly weakens the model's grounding ability, leading to incorrect predictions and drastic performance degradation. In Referring Expression Comprehension (REC), for instance, pruning causes the accuracy of L"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.21873","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.21873/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.21873","created_at":"2026-07-05T11:28:03.108000+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.21873v1","created_at":"2026-07-05T11:28:03.108000+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.21873","created_at":"2026-07-05T11:28:03.108000+00:00"},{"alias_kind":"pith_short_12","alias_value":"CMCU2KGTIEBA","created_at":"2026-07-05T11:28:03.108000+00:00"},{"alias_kind":"pith_short_16","alias_value":"CMCU2KGTIEBAF3PU","created_at":"2026-07-05T11:28:03.108000+00:00"},{"alias_kind":"pith_short_8","alias_value":"CMCU2KGT","created_at":"2026-07-05T11:28:03.108000+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12412","citing_title":"Reroute, Don't Remove: Recoverable Visual Token Routing for Vision-Language Models","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31599","citing_title":"Token-Sparse Medical Multimodal Reasoning via Dual-Stream Reinforcement Learning","ref_index":26,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CMCU2KGTIEBAF3PUGSAIT5KISN","json":"https://pith.science/pith/CMCU2KGTIEBAF3PUGSAIT5KISN.json","graph_json":"https://pith.science/api/pith-number/CMCU2KGTIEBAF3PUGSAIT5KISN/graph.json","events_json":"https://pith.science/api/pith-number/CMCU2KGTIEBAF3PUGSAIT5KISN/events.json","paper":"https://pith.science/paper/CMCU2KGT"},"agent_actions":{"view_html":"https://pith.science/pith/CMCU2KGTIEBAF3PUGSAIT5KISN","download_json":"https://pith.science/pith/CMCU2KGTIEBAF3PUGSAIT5KISN.json","view_paper":"https://pith.science/paper/CMCU2KGT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.21873&json=true","fetch_graph":"https://pith.science/api/pith-number/CMCU2KGTIEBAF3PUGSAIT5KISN/graph.json","fetch_events":"https://pith.science/api/pith-number/CMCU2KGTIEBAF3PUGSAIT5KISN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CMCU2KGTIEBAF3PUGSAIT5KISN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CMCU2KGTIEBAF3PUGSAIT5KISN/action/storage_attestation","attest_author":"https://pith.science/pith/CMCU2KGTIEBAF3PUGSAIT5KISN/action/author_attestation","sign_citation":"https://pith.science/pith/CMCU2KGTIEBAF3PUGSAIT5KISN/action/citation_signature","submit_replication":"https://pith.science/pith/CMCU2KGTIEBAF3PUGSAIT5KISN/action/replication_record"}},"created_at":"2026-07-05T11:28:03.108000+00:00","updated_at":"2026-07-05T11:28:03.108000+00:00"}