{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:WXJFORNDSKFA42J2ZT6V3OTAI2","short_pith_number":"pith:WXJFORND","schema_version":"1.0","canonical_sha256":"b5d25745a3928a0e693accfd5dba6046a7271c97f2a2c13c293c11eb1437031a","source":{"kind":"arxiv","id":"2503.10501","version":1},"attestation_state":"computed","paper":{"title":"TokenCarve: Information-Preserving Visual Token Compression in Multimodal Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chongjun Tu, Dongzhan Zhou, Jianjian Cao, Lin Zhang, Peng Ye, Tao Chen, Xudong Tan, Yaoxin Yang","submitted_at":"2025-03-13T16:04:31Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) are becoming increasingly popular, while the high computational cost associated with multimodal data input, particularly from visual tokens, poses a significant challenge. Existing training-based token compression methods improve inference efficiency but require costly retraining, while training-free methods struggle to maintain performance when aggressively reducing token counts. In this study, we reveal that the performance degradation of MLLM closely correlates with the accelerated loss of information in the attention output matrix. This insight intr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.10501","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-03-13T16:04:31Z","cross_cats_sorted":[],"title_canon_sha256":"da7685e031bd6e4bf14493c2c5e41c228cd1aa8157ac81ebd78f3ab02c3ca16e","abstract_canon_sha256":"e3ad3789bfd793924eebf84851aedb42c327bb62b2ef5dd0c3f197701daec072"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:30:46.819660Z","signature_b64":"qbYRbQde0TfitMA+th6KsH+yrXa6MZfImK8YyKGhoNntfOC4P6l069hO8Z7lJOyN4UI4eh+CIuweNK7GEJY/Ag==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b5d25745a3928a0e693accfd5dba6046a7271c97f2a2c13c293c11eb1437031a","last_reissued_at":"2026-07-05T10:30:46.819022Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:30:46.819022Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TokenCarve: Information-Preserving Visual Token Compression in Multimodal Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chongjun Tu, Dongzhan Zhou, Jianjian Cao, Lin Zhang, Peng Ye, Tao Chen, Xudong Tan, Yaoxin Yang","submitted_at":"2025-03-13T16:04:31Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) are becoming increasingly popular, while the high computational cost associated with multimodal data input, particularly from visual tokens, poses a significant challenge. Existing training-based token compression methods improve inference efficiency but require costly retraining, while training-free methods struggle to maintain performance when aggressively reducing token counts. In this study, we reveal that the performance degradation of MLLM closely correlates with the accelerated loss of information in the attention output matrix. This insight intr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.10501","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.10501/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.10501","created_at":"2026-07-05T10:30:46.819095+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.10501v1","created_at":"2026-07-05T10:30:46.819095+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.10501","created_at":"2026-07-05T10:30:46.819095+00:00"},{"alias_kind":"pith_short_12","alias_value":"WXJFORNDSKFA","created_at":"2026-07-05T10:30:46.819095+00:00"},{"alias_kind":"pith_short_16","alias_value":"WXJFORNDSKFA42J2","created_at":"2026-07-05T10:30:46.819095+00:00"},{"alias_kind":"pith_short_8","alias_value":"WXJFORND","created_at":"2026-07-05T10:30:46.819095+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.02842","citing_title":"Spectral-Progressive Thought Flow for Lightweight Multimodal Reasoning","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2511.14582","citing_title":"OmniZip: Audio-Guided Dynamic Token Compression for Fast Omnimodal Large Language Models","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2603.01400","citing_title":"Token Reduction via Local and Global Contexts Optimization for Efficient Video Large Language Models","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12056","citing_title":"OmniRefine: Alignment-Aware Cooperative Compression for Efficient Omnimodal Large Language Models","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11605","citing_title":"Keep What Audio Cannot Say: Context-Preserving Token Pruning for Omni-LLMs","ref_index":44,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WXJFORNDSKFA42J2ZT6V3OTAI2","json":"https://pith.science/pith/WXJFORNDSKFA42J2ZT6V3OTAI2.json","graph_json":"https://pith.science/api/pith-number/WXJFORNDSKFA42J2ZT6V3OTAI2/graph.json","events_json":"https://pith.science/api/pith-number/WXJFORNDSKFA42J2ZT6V3OTAI2/events.json","paper":"https://pith.science/paper/WXJFORND"},"agent_actions":{"view_html":"https://pith.science/pith/WXJFORNDSKFA42J2ZT6V3OTAI2","download_json":"https://pith.science/pith/WXJFORNDSKFA42J2ZT6V3OTAI2.json","view_paper":"https://pith.science/paper/WXJFORND","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.10501&json=true","fetch_graph":"https://pith.science/api/pith-number/WXJFORNDSKFA42J2ZT6V3OTAI2/graph.json","fetch_events":"https://pith.science/api/pith-number/WXJFORNDSKFA42J2ZT6V3OTAI2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WXJFORNDSKFA42J2ZT6V3OTAI2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WXJFORNDSKFA42J2ZT6V3OTAI2/action/storage_attestation","attest_author":"https://pith.science/pith/WXJFORNDSKFA42J2ZT6V3OTAI2/action/author_attestation","sign_citation":"https://pith.science/pith/WXJFORNDSKFA42J2ZT6V3OTAI2/action/citation_signature","submit_replication":"https://pith.science/pith/WXJFORNDSKFA42J2ZT6V3OTAI2/action/replication_record"}},"created_at":"2026-07-05T10:30:46.819095+00:00","updated_at":"2026-07-05T10:30:46.819095+00:00"}