{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XD7UMR2TJPM5LTGTCMQ3H3GVNO","short_pith_number":"pith:XD7UMR2T","schema_version":"1.0","canonical_sha256":"b8ff4647534bd9d5ccd31321b3ecd56bb5c3ca5b338af32226469059e1bc1b1b","source":{"kind":"arxiv","id":"2408.10945","version":3},"attestation_state":"computed","paper":{"title":"HiRED: Attention-Guided Token Dropping for Efficient Inference of High-Resolution Vision-Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Bo Ji, Deepu John, Dimitrios S. Nikolopoulos, Hans Vandierendonck, JinYi Yoon, Kazi Hasan Ibn Arif","submitted_at":"2024-08-20T15:34:27Z","abstract_excerpt":"High-resolution Vision-Language Models (VLMs) are widely used in multimodal tasks to enhance accuracy by preserving detailed image information. However, these models often generate an excessive number of visual tokens due to the need to encode multiple partitions of a high-resolution image input. Processing such a large number of visual tokens through multiple transformer networks poses significant computational challenges, particularly for resource-constrained commodity GPUs. To address this challenge, we propose High-Resolution Early Dropping (HiRED), a plug-and-play token-dropping method de"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.10945","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-08-20T15:34:27Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"0dde21a2b756ac8ac52c2db41b6ff5d3d5a693f5856f23ce65af53461ad65c3f","abstract_canon_sha256":"c9c3814d6badd082e2ac0fbb61edb4589490f622cf967f47904ea0cbc93ac068"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:54:03.756195Z","signature_b64":"yAiLfIiOljBITjhMuHf81D2OXrK4quWYSd11MH+0wUkJNzBpJLh0c+noSf7TkEBEdn3S7lm5I6/ihufampV2CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b8ff4647534bd9d5ccd31321b3ecd56bb5c3ca5b338af32226469059e1bc1b1b","last_reissued_at":"2026-07-05T09:54:03.755707Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:54:03.755707Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"HiRED: Attention-Guided Token Dropping for Efficient Inference of High-Resolution Vision-Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Bo Ji, Deepu John, Dimitrios S. Nikolopoulos, Hans Vandierendonck, JinYi Yoon, Kazi Hasan Ibn Arif","submitted_at":"2024-08-20T15:34:27Z","abstract_excerpt":"High-resolution Vision-Language Models (VLMs) are widely used in multimodal tasks to enhance accuracy by preserving detailed image information. However, these models often generate an excessive number of visual tokens due to the need to encode multiple partitions of a high-resolution image input. Processing such a large number of visual tokens through multiple transformer networks poses significant computational challenges, particularly for resource-constrained commodity GPUs. To address this challenge, we propose High-Resolution Early Dropping (HiRED), a plug-and-play token-dropping method de"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.10945","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.10945/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.10945","created_at":"2026-07-05T09:54:03.755768+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.10945v3","created_at":"2026-07-05T09:54:03.755768+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.10945","created_at":"2026-07-05T09:54:03.755768+00:00"},{"alias_kind":"pith_short_12","alias_value":"XD7UMR2TJPM5","created_at":"2026-07-05T09:54:03.755768+00:00"},{"alias_kind":"pith_short_16","alias_value":"XD7UMR2TJPM5LTGT","created_at":"2026-07-05T09:54:03.755768+00:00"},{"alias_kind":"pith_short_8","alias_value":"XD7UMR2T","created_at":"2026-07-05T09:54:03.755768+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2410.17247","citing_title":"PyramidDrop: Accelerating Your Large Vision-Language Models via Pyramid Visual Redundancy Reduction","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01048","citing_title":"Compared to What? Baselines and Metrics for Counterfactual Prompting","ref_index":86,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XD7UMR2TJPM5LTGTCMQ3H3GVNO","json":"https://pith.science/pith/XD7UMR2TJPM5LTGTCMQ3H3GVNO.json","graph_json":"https://pith.science/api/pith-number/XD7UMR2TJPM5LTGTCMQ3H3GVNO/graph.json","events_json":"https://pith.science/api/pith-number/XD7UMR2TJPM5LTGTCMQ3H3GVNO/events.json","paper":"https://pith.science/paper/XD7UMR2T"},"agent_actions":{"view_html":"https://pith.science/pith/XD7UMR2TJPM5LTGTCMQ3H3GVNO","download_json":"https://pith.science/pith/XD7UMR2TJPM5LTGTCMQ3H3GVNO.json","view_paper":"https://pith.science/paper/XD7UMR2T","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.10945&json=true","fetch_graph":"https://pith.science/api/pith-number/XD7UMR2TJPM5LTGTCMQ3H3GVNO/graph.json","fetch_events":"https://pith.science/api/pith-number/XD7UMR2TJPM5LTGTCMQ3H3GVNO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XD7UMR2TJPM5LTGTCMQ3H3GVNO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XD7UMR2TJPM5LTGTCMQ3H3GVNO/action/storage_attestation","attest_author":"https://pith.science/pith/XD7UMR2TJPM5LTGTCMQ3H3GVNO/action/author_attestation","sign_citation":"https://pith.science/pith/XD7UMR2TJPM5LTGTCMQ3H3GVNO/action/citation_signature","submit_replication":"https://pith.science/pith/XD7UMR2TJPM5LTGTCMQ3H3GVNO/action/replication_record"}},"created_at":"2026-07-05T09:54:03.755768+00:00","updated_at":"2026-07-05T09:54:03.755768+00:00"}