{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:Z2QMCJLCZWB2V66V2EKKRQHHPE","short_pith_number":"pith:Z2QMCJLC","schema_version":"1.0","canonical_sha256":"cea0c12562cd83aafbd5d114a8c0e7791ce865320a5d150d4a9603ddfb44d810","source":{"kind":"arxiv","id":"2412.01818","version":2},"attestation_state":"computed","paper":{"title":"Beyond Text-Visual Attention: Exploiting Visual Cues for Effective Token Pruning in VLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Aosong Cheng, Jiajun Cao, Ming Lu, Qi She, Qizhe Zhang, Renrui Zhang, Shanghang Zhang, Shaobo Guo, Zhiyong Zhuo","submitted_at":"2024-12-02T18:57:40Z","abstract_excerpt":"Large vision-language models (LVLMs) generally contain significantly more visual tokens than their textual counterparts, resulting in a considerable computational burden. Recent efforts have been made to tackle this issue by pruning visual tokens early within the language model. Most existing works use attention scores between text and visual tokens to assess the importance of visual tokens. However, in this study, we first analyze the text-visual attention in the language model and find that this score is not an ideal indicator for token pruning. Based on the analysis, We propose VisPruner, a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.01818","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-12-02T18:57:40Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"0344ecc6c44ffb602d0707a16e4e848ff75007aee0aa99f8876b403ce7133a31","abstract_canon_sha256":"067565c7f64791952612cc1479d90f15b67b928fac3e8ae2e00da4c2872ab4c7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:01:29.491711Z","signature_b64":"C9nPKnL8ytMXKyM5UOH83LSklAswUdRPgm5qOCZdmbFlQYADR4C4iuTzHlG7MQ2ydvYFvXdpZ9daQjMFKILhCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cea0c12562cd83aafbd5d114a8c0e7791ce865320a5d150d4a9603ddfb44d810","last_reissued_at":"2026-07-05T11:01:29.491217Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:01:29.491217Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Beyond Text-Visual Attention: Exploiting Visual Cues for Effective Token Pruning in VLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Aosong Cheng, Jiajun Cao, Ming Lu, Qi She, Qizhe Zhang, Renrui Zhang, Shanghang Zhang, Shaobo Guo, Zhiyong Zhuo","submitted_at":"2024-12-02T18:57:40Z","abstract_excerpt":"Large vision-language models (LVLMs) generally contain significantly more visual tokens than their textual counterparts, resulting in a considerable computational burden. Recent efforts have been made to tackle this issue by pruning visual tokens early within the language model. Most existing works use attention scores between text and visual tokens to assess the importance of visual tokens. However, in this study, we first analyze the text-visual attention in the language model and find that this score is not an ideal indicator for token pruning. Based on the analysis, We propose VisPruner, a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.01818","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.01818/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.01818","created_at":"2026-07-05T11:01:29.491275+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.01818v2","created_at":"2026-07-05T11:01:29.491275+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.01818","created_at":"2026-07-05T11:01:29.491275+00:00"},{"alias_kind":"pith_short_12","alias_value":"Z2QMCJLCZWB2","created_at":"2026-07-05T11:01:29.491275+00:00"},{"alias_kind":"pith_short_16","alias_value":"Z2QMCJLCZWB2V66V","created_at":"2026-07-05T11:01:29.491275+00:00"},{"alias_kind":"pith_short_8","alias_value":"Z2QMCJLC","created_at":"2026-07-05T11:01:29.491275+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.08511","citing_title":"Look Less, Reason More: Block-wise Attention Skipping for Efficient Multimodal LLMs","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31383","citing_title":"MS-Resampler: Multi-Scope Visual Resampling for Efficient Multimodal LLMs","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25179","citing_title":"Locality Matters for Training-Free Audio Token Compression in Audio-Language Models","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2503.14075","citing_title":"Growing a Multi-head Twig via Distillation and Reinforcement Learning to Accelerate Large Vision-Language Models","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15621","citing_title":"LRCP: Low-Rank Compressibility Guided Visual Token Pruning for Efficient LVLMs","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09429","citing_title":"Evading Visual Aphasia: Contrastive Adaptive Semantic Token Pruning for Vision-Language Models","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09982","citing_title":"ERASE: Eliminating Redundant Visual Tokens via Adaptive Two-Stage Token Pruning","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11122","citing_title":"Semantic-Geometric Dual Compression: Training-Free Visual Token Reduction for Ultra-High-Resolution Remote Sensing Understanding","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05601","citing_title":"ID-Selection: Importance-Diversity Based Visual Token Selection for Efficient LVLM Inference","ref_index":40,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Z2QMCJLCZWB2V66V2EKKRQHHPE","json":"https://pith.science/pith/Z2QMCJLCZWB2V66V2EKKRQHHPE.json","graph_json":"https://pith.science/api/pith-number/Z2QMCJLCZWB2V66V2EKKRQHHPE/graph.json","events_json":"https://pith.science/api/pith-number/Z2QMCJLCZWB2V66V2EKKRQHHPE/events.json","paper":"https://pith.science/paper/Z2QMCJLC"},"agent_actions":{"view_html":"https://pith.science/pith/Z2QMCJLCZWB2V66V2EKKRQHHPE","download_json":"https://pith.science/pith/Z2QMCJLCZWB2V66V2EKKRQHHPE.json","view_paper":"https://pith.science/paper/Z2QMCJLC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.01818&json=true","fetch_graph":"https://pith.science/api/pith-number/Z2QMCJLCZWB2V66V2EKKRQHHPE/graph.json","fetch_events":"https://pith.science/api/pith-number/Z2QMCJLCZWB2V66V2EKKRQHHPE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Z2QMCJLCZWB2V66V2EKKRQHHPE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Z2QMCJLCZWB2V66V2EKKRQHHPE/action/storage_attestation","attest_author":"https://pith.science/pith/Z2QMCJLCZWB2V66V2EKKRQHHPE/action/author_attestation","sign_citation":"https://pith.science/pith/Z2QMCJLCZWB2V66V2EKKRQHHPE/action/citation_signature","submit_replication":"https://pith.science/pith/Z2QMCJLCZWB2V66V2EKKRQHHPE/action/replication_record"}},"created_at":"2026-07-05T11:01:29.491275+00:00","updated_at":"2026-07-05T11:01:29.491275+00:00"}