{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:EP5U3VZB7IEX32H5E5UJYBI7FG","short_pith_number":"pith:EP5U3VZB","schema_version":"1.0","canonical_sha256":"23fb4dd721fa097de8fd27689c051f29a68e7cca53aa0844cdf6705f2ed9f3ec","source":{"kind":"arxiv","id":"2503.15816","version":3},"attestation_state":"computed","paper":{"title":"A Vision Centric Remote Sensing Benchmark","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Abduljaleel Adejumo, Aoran Xiao, Clifford Broni-Bediako, Faegheh Yeganli, Mennatullah Siam, Naoto Yokoya","submitted_at":"2025-03-20T03:03:46Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) have achieved remarkable success in vision-language tasks but their remote sensing (RS) counterpart are relatively under explored. Unlike natural images, RS imagery presents unique challenges that current MLLMs struggle to handle, particularly in visual grounding and spatial reasoning. This study investigates the limitations of CLIP-based MLLMs in RS, highlighting their failure to differentiate visually distinct yet semantically similar RS images. To address this, we introduce a remote sensing multimodal visual patterns (RSMMVP) benchmark. It is designe"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.15816","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-03-20T03:03:46Z","cross_cats_sorted":[],"title_canon_sha256":"138e36d5d5b64d5bc23e7a92f0030dc80a764e12bdfb944bc9b75e3db6103aa9","abstract_canon_sha256":"033bb69c452e6df6106d426c273d66d550b760ccb791ce91abc24f4c018e12c6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:01:03.967960Z","signature_b64":"tGAaK7hKrW8ZB7hLD/bMacZXuwPAxlg8m3Jmkn6/jnzNdr3kCCs3fXyJjcgAbgLEQALVK3v4NIjv6grxdlUzCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"23fb4dd721fa097de8fd27689c051f29a68e7cca53aa0844cdf6705f2ed9f3ec","last_reissued_at":"2026-07-05T11:01:03.967379Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:01:03.967379Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Vision Centric Remote Sensing Benchmark","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Abduljaleel Adejumo, Aoran Xiao, Clifford Broni-Bediako, Faegheh Yeganli, Mennatullah Siam, Naoto Yokoya","submitted_at":"2025-03-20T03:03:46Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) have achieved remarkable success in vision-language tasks but their remote sensing (RS) counterpart are relatively under explored. Unlike natural images, RS imagery presents unique challenges that current MLLMs struggle to handle, particularly in visual grounding and spatial reasoning. This study investigates the limitations of CLIP-based MLLMs in RS, highlighting their failure to differentiate visually distinct yet semantically similar RS images. To address this, we introduce a remote sensing multimodal visual patterns (RSMMVP) benchmark. It is designe"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.15816","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.15816/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.15816","created_at":"2026-07-05T11:01:03.967437+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.15816v3","created_at":"2026-07-05T11:01:03.967437+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.15816","created_at":"2026-07-05T11:01:03.967437+00:00"},{"alias_kind":"pith_short_12","alias_value":"EP5U3VZB7IEX","created_at":"2026-07-05T11:01:03.967437+00:00"},{"alias_kind":"pith_short_16","alias_value":"EP5U3VZB7IEX32H5","created_at":"2026-07-05T11:01:03.967437+00:00"},{"alias_kind":"pith_short_8","alias_value":"EP5U3VZB","created_at":"2026-07-05T11:01:03.967437+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EP5U3VZB7IEX32H5E5UJYBI7FG","json":"https://pith.science/pith/EP5U3VZB7IEX32H5E5UJYBI7FG.json","graph_json":"https://pith.science/api/pith-number/EP5U3VZB7IEX32H5E5UJYBI7FG/graph.json","events_json":"https://pith.science/api/pith-number/EP5U3VZB7IEX32H5E5UJYBI7FG/events.json","paper":"https://pith.science/paper/EP5U3VZB"},"agent_actions":{"view_html":"https://pith.science/pith/EP5U3VZB7IEX32H5E5UJYBI7FG","download_json":"https://pith.science/pith/EP5U3VZB7IEX32H5E5UJYBI7FG.json","view_paper":"https://pith.science/paper/EP5U3VZB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.15816&json=true","fetch_graph":"https://pith.science/api/pith-number/EP5U3VZB7IEX32H5E5UJYBI7FG/graph.json","fetch_events":"https://pith.science/api/pith-number/EP5U3VZB7IEX32H5E5UJYBI7FG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EP5U3VZB7IEX32H5E5UJYBI7FG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EP5U3VZB7IEX32H5E5UJYBI7FG/action/storage_attestation","attest_author":"https://pith.science/pith/EP5U3VZB7IEX32H5E5UJYBI7FG/action/author_attestation","sign_citation":"https://pith.science/pith/EP5U3VZB7IEX32H5E5UJYBI7FG/action/citation_signature","submit_replication":"https://pith.science/pith/EP5U3VZB7IEX32H5E5UJYBI7FG/action/replication_record"}},"created_at":"2026-07-05T11:01:03.967437+00:00","updated_at":"2026-07-05T11:01:03.967437+00:00"}