{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:MNW47ZMGTUITETHOS2ZXVPE6PW","short_pith_number":"pith:MNW47ZMG","schema_version":"1.0","canonical_sha256":"636dcfe5869d11324cee96b37abc9e7d9ab5e98b553395002ad8069c02325df7","source":{"kind":"arxiv","id":"2503.07588","version":3},"attestation_state":"computed","paper":{"title":"When Large Vision-Language Model Meets Large Remote Sensing Imagery: Coarse-to-Fine Text-Guided Token Pruning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Jingdong Chen, Junwei Luo, Kang Wu, Lei Liang, Qi Zhu, Xue Yang, Yansheng Li, Yingying Zhang","submitted_at":"2025-03-10T17:51:16Z","abstract_excerpt":"Efficient vision-language understanding of large Remote Sensing Images (RSIs) is meaningful but challenging. Current Large Vision-Language Models (LVLMs) typically employ limited pre-defined grids to process images, leading to information loss when handling gigapixel RSIs. Conversely, using unlimited grids significantly increases computational costs. To preserve image details while reducing computational complexity, we propose a text-guided token pruning method with Dynamic Image Pyramid (DIP) integration. Our method introduces: (i) a Region Focus Module (RFM) that leverages text-aware region "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.07588","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-03-10T17:51:16Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"4f3ae3c71121edf826343d936ed491c45a0dc8cfa76842d20357a869fb1fb704","abstract_canon_sha256":"4e38348c7518e139fa255d4d4b859ea02fc8252ba330206b92267371126b350b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:42:27.009093Z","signature_b64":"BY/7vAB8I9P28/X24VNc2jZ8fbJL5wgW05prHDVMeUrcTFyrx0rqhv9l3K0VFbprAEsejn0YtIHleJArGcofDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"636dcfe5869d11324cee96b37abc9e7d9ab5e98b553395002ad8069c02325df7","last_reissued_at":"2026-07-05T11:42:27.008611Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:42:27.008611Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"When Large Vision-Language Model Meets Large Remote Sensing Imagery: Coarse-to-Fine Text-Guided Token Pruning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Jingdong Chen, Junwei Luo, Kang Wu, Lei Liang, Qi Zhu, Xue Yang, Yansheng Li, Yingying Zhang","submitted_at":"2025-03-10T17:51:16Z","abstract_excerpt":"Efficient vision-language understanding of large Remote Sensing Images (RSIs) is meaningful but challenging. Current Large Vision-Language Models (LVLMs) typically employ limited pre-defined grids to process images, leading to information loss when handling gigapixel RSIs. Conversely, using unlimited grids significantly increases computational costs. To preserve image details while reducing computational complexity, we propose a text-guided token pruning method with Dynamic Image Pyramid (DIP) integration. Our method introduces: (i) a Region Focus Module (RFM) that leverages text-aware region "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.07588","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.07588/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.07588","created_at":"2026-07-05T11:42:27.008671+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.07588v3","created_at":"2026-07-05T11:42:27.008671+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.07588","created_at":"2026-07-05T11:42:27.008671+00:00"},{"alias_kind":"pith_short_12","alias_value":"MNW47ZMGTUIT","created_at":"2026-07-05T11:42:27.008671+00:00"},{"alias_kind":"pith_short_16","alias_value":"MNW47ZMGTUITETHO","created_at":"2026-07-05T11:42:27.008671+00:00"},{"alias_kind":"pith_short_8","alias_value":"MNW47ZMG","created_at":"2026-07-05T11:42:27.008671+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.14475","citing_title":"GeoVista: Visually Grounded Active Perception for Ultra-High-Resolution Remote Sensing Understanding","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MNW47ZMGTUITETHOS2ZXVPE6PW","json":"https://pith.science/pith/MNW47ZMGTUITETHOS2ZXVPE6PW.json","graph_json":"https://pith.science/api/pith-number/MNW47ZMGTUITETHOS2ZXVPE6PW/graph.json","events_json":"https://pith.science/api/pith-number/MNW47ZMGTUITETHOS2ZXVPE6PW/events.json","paper":"https://pith.science/paper/MNW47ZMG"},"agent_actions":{"view_html":"https://pith.science/pith/MNW47ZMGTUITETHOS2ZXVPE6PW","download_json":"https://pith.science/pith/MNW47ZMGTUITETHOS2ZXVPE6PW.json","view_paper":"https://pith.science/paper/MNW47ZMG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.07588&json=true","fetch_graph":"https://pith.science/api/pith-number/MNW47ZMGTUITETHOS2ZXVPE6PW/graph.json","fetch_events":"https://pith.science/api/pith-number/MNW47ZMGTUITETHOS2ZXVPE6PW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MNW47ZMGTUITETHOS2ZXVPE6PW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MNW47ZMGTUITETHOS2ZXVPE6PW/action/storage_attestation","attest_author":"https://pith.science/pith/MNW47ZMGTUITETHOS2ZXVPE6PW/action/author_attestation","sign_citation":"https://pith.science/pith/MNW47ZMGTUITETHOS2ZXVPE6PW/action/citation_signature","submit_replication":"https://pith.science/pith/MNW47ZMGTUITETHOS2ZXVPE6PW/action/replication_record"}},"created_at":"2026-07-05T11:42:27.008671+00:00","updated_at":"2026-07-05T11:42:27.008671+00:00"}