{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:QUSHR6WH544A2PINBG53GSRP6Q","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"9c82ac07718480edb4405202e8eea1d3e1767846b96faaf68464ebc7e84c35a9","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2026-05-13T08:40:40Z","title_canon_sha256":"3723d4e9b36ff76cf8d891b08b3348d7cf26c77e5522988302b02f7ec78f7e66"},"schema_version":"1.0","source":{"id":"2605.13178","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2605.13178","created_at":"2026-05-18T03:08:56Z"},{"alias_kind":"arxiv_version","alias_value":"2605.13178v1","created_at":"2026-05-18T03:08:56Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2605.13178","created_at":"2026-05-18T03:08:56Z"},{"alias_kind":"pith_short_12","alias_value":"QUSHR6WH544A","created_at":"2026-05-18T12:33:37Z"},{"alias_kind":"pith_short_16","alias_value":"QUSHR6WH544A2PIN","created_at":"2026-05-18T12:33:37Z"},{"alias_kind":"pith_short_8","alias_value":"QUSHR6WH","created_at":"2026-05-18T12:33:37Z"}],"graph_snapshots":[{"event_id":"sha256:a84485b0902e03c59738a2c6444a12de19c841796da924f361f48eb28d9f2a8d","target":"graph","created_at":"2026-05-18T03:08:56Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"LiteLVLM significantly outperforms existing methods by over 5% across diverse token budgets. Without any training or fine-tuning, LiteLVLM maintains 90% of the original performance with a 22% speedup and a 2.3x memory reduction."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"The observation that referent-region visual tokens exhibit low similarity to text in CLIP analysis generalizes directly to the large vision-language models used for pixel grounding, and that reversing the similarity ranking will reliably retain the necessary tokens across inputs and models."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"LiteLVLM prunes visual tokens for pixel grounding by reversing CLIP visual-text similarity to retain referent region tokens, outperforming prior methods by over 5% with 22% speedup and 2.3x memory reduction without any training."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"Reversing CLIP visual-text similarity retains the tokens needed for accurate pixel grounding without training."}],"snapshot_sha256":"acd0ca8d3c49cc20b2c5de6b6d41f6a88a8a271e52fc9f1392823fcda969bf13"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"paper":{"abstract_excerpt":"In large vision-language models, visual tokens typically constitute the majority of input tokens, leading to substantial computational overhead. To address this, recent studies have explored pruning redundant or less informative visual tokens for image understanding tasks. However, these methods struggle with pixel grounding tasks, where token importance is highly contingent on the input text. Through an in-depth analysis of CLIP, we observe that visual tokens located within referent regions often exhibit low similarity to the textual representation. Motivated by this insight, we introduce Lit","authors_text":"Sangin Lee, Yukyung Choi","cross_cats":["cs.AI"],"headline":"Reversing CLIP visual-text similarity retains the tokens needed for accurate pixel grounding without training.","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2026-05-13T08:40:40Z","title":"CLIP Tricks You: Training-free Token Pruning for Efficient Pixel Grounding in Large VIsion-Language Models"},"references":{"count":15,"internal_anchors":8,"resolved_work":15,"sample":[{"cited_arxiv_id":"2303.08774","doi":"","is_internal_anchor":true,"ref_index":1,"title":"GPT-4 Technical Report","work_id":"b928e041-6991-4c08-8c81-0359e4097c7b","year":null},{"cited_arxiv_id":"2309.16609","doi":"","is_internal_anchor":true,"ref_index":2,"title":"Qwen Technical Report","work_id":"bb1fd52f-6b2f-437c-9516-37bdf6eb9be8","year":null},{"cited_arxiv_id":"2511.21631","doi":"","is_internal_anchor":true,"ref_index":3,"title":"Qwen3-VL Technical Report","work_id":"1fe243aa-e3c0-4da6-b391-4cbcfc88d5c0","year":null},{"cited_arxiv_id":"2312.14125","doi":"","is_internal_anchor":true,"ref_index":4,"title":"VideoPoet: A Large Language Model for Zero-Shot Video Generation","work_id":"5cc3572d-7e2f-4431-ae42-d9282a42a800","year":null},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":5,"title":"Liu, H., Li, C., Li, Y ., and Lee, Y . J. Improved baselines with visual instruction tuning. InProceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition, 2024a. Liu, H., Li, C.","work_id":"6f376968-7f61-495c-a4fd-17618ac7797f","year":2024}],"snapshot_sha256":"f61f83705ee40fde067ab55ddeea7103d5f8baebbfe50d66aed4348815b8370b"},"source":{"id":"2605.13178","kind":"arxiv","version":1},"verdict":{"created_at":"2026-05-14T20:34:10.754156Z","id":"da1817eb-e7d1-4a07-8f49-17e12627316a","model_set":{"reader":"grok-4.3"},"one_line_summary":"LiteLVLM prunes visual tokens for pixel grounding by reversing CLIP visual-text similarity to retain referent region tokens, outperforming prior methods by over 5% with 22% speedup and 2.3x memory reduction without any training.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"Reversing CLIP visual-text similarity retains the tokens needed for accurate pixel grounding without training.","strongest_claim":"LiteLVLM significantly outperforms existing methods by over 5% across diverse token budgets. Without any training or fine-tuning, LiteLVLM maintains 90% of the original performance with a 22% speedup and a 2.3x memory reduction.","weakest_assumption":"The observation that referent-region visual tokens exhibit low similarity to text in CLIP analysis generalizes directly to the large vision-language models used for pixel grounding, and that reversing the similarity ranking will reliably retain the necessary tokens across inputs and models."}},"verdict_id":"da1817eb-e7d1-4a07-8f49-17e12627316a"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:f8ee8bfafcf6f954f6e765162f5dac0646feb09b7c7469857d4543d48916a246","target":"record","created_at":"2026-05-18T03:08:56Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"9c82ac07718480edb4405202e8eea1d3e1767846b96faaf68464ebc7e84c35a9","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2026-05-13T08:40:40Z","title_canon_sha256":"3723d4e9b36ff76cf8d891b08b3348d7cf26c77e5522988302b02f7ec78f7e66"},"schema_version":"1.0","source":{"id":"2605.13178","kind":"arxiv","version":1}},"canonical_sha256":"852478fac7ef380d3d0d09bbb34a2ff4131e3b3a4dee98b0e4e64acf152cb334","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"852478fac7ef380d3d0d09bbb34a2ff4131e3b3a4dee98b0e4e64acf152cb334","first_computed_at":"2026-05-18T03:08:56.450163Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-05-18T03:08:56.450163Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"80Ffx6s+XDe3/tfUNB6eWkeobkAcctHlzQORw6sZHPkjYyL+6MWdRUcm1Rkf5vB/TBhP/1WcnbLaTGTRKZz7CQ==","signature_status":"signed_v1","signed_at":"2026-05-18T03:08:56.450668Z","signed_message":"canonical_sha256_bytes"},"source_id":"2605.13178","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:f8ee8bfafcf6f954f6e765162f5dac0646feb09b7c7469857d4543d48916a246","sha256:a84485b0902e03c59738a2c6444a12de19c841796da924f361f48eb28d9f2a8d"],"state_sha256":"60626e369baf67f5b0bc88760a464ee2b2f55da0f1b7c8393e038302cb993ebe"}