{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:NNNS6ZVMF3BOZKANOWBOCYFCVA","short_pith_number":"pith:NNNS6ZVM","schema_version":"1.0","canonical_sha256":"6b5b2f66ac2ec2eca80d7582e160a2a83519d632e0afff08d6684ad263aa33b8","source":{"kind":"arxiv","id":"2503.23771","version":1},"attestation_state":"computed","paper":{"title":"XLRS-Bench: Could Your Multimodal LLMs Understand Extremely Large Ultra-High-Resolution Remote Sensing Imagery?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Di Wang, Fengxiang Wang, Hongzhen Wang, Jing Zhang, Long Lan, Maosong Sun, Mingshuo Chen, Qiang Ma, Wenjing Yang, Yulin Wang, Zhiyuan Liu, Zonghao Guo","submitted_at":"2025-03-31T06:41:18Z","abstract_excerpt":"The astonishing breakthrough of multimodal large language models (MLLMs) has necessitated new benchmarks to quantitatively assess their capabilities, reveal their limitations, and indicate future research directions. However, this is challenging in the context of remote sensing (RS), since the imagery features ultra-high resolution that incorporates extremely complex semantic relationships. Existing benchmarks usually adopt notably smaller image sizes than real-world RS scenarios, suffer from limited annotation quality, and consider insufficient dimensions of evaluation. To address these issue"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.23771","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-03-31T06:41:18Z","cross_cats_sorted":[],"title_canon_sha256":"a5cc236bb52c34bbff73656b3094fdd91c25728fc1ad3f84794a6890b0c1f2fd","abstract_canon_sha256":"52f2e8198395f3f6333717d234b120ebbbe64de8d5a5026322e41450b27d76df"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:41:59.664095Z","signature_b64":"jfWFx4+2dxvk52boFlTG7iFHpu3BT+eyX8GTRNOjYYOZZJXiFeMOiYj/GD+i+cpOyB2lLUTAaAwIDCsbXl8RCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6b5b2f66ac2ec2eca80d7582e160a2a83519d632e0afff08d6684ad263aa33b8","last_reissued_at":"2026-07-05T10:41:59.663587Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:41:59.663587Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"XLRS-Bench: Could Your Multimodal LLMs Understand Extremely Large Ultra-High-Resolution Remote Sensing Imagery?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Di Wang, Fengxiang Wang, Hongzhen Wang, Jing Zhang, Long Lan, Maosong Sun, Mingshuo Chen, Qiang Ma, Wenjing Yang, Yulin Wang, Zhiyuan Liu, Zonghao Guo","submitted_at":"2025-03-31T06:41:18Z","abstract_excerpt":"The astonishing breakthrough of multimodal large language models (MLLMs) has necessitated new benchmarks to quantitatively assess their capabilities, reveal their limitations, and indicate future research directions. However, this is challenging in the context of remote sensing (RS), since the imagery features ultra-high resolution that incorporates extremely complex semantic relationships. Existing benchmarks usually adopt notably smaller image sizes than real-world RS scenarios, suffer from limited annotation quality, and consider insufficient dimensions of evaluation. To address these issue"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.23771","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.23771/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.23771","created_at":"2026-07-05T10:41:59.663647+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.23771v1","created_at":"2026-07-05T10:41:59.663647+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.23771","created_at":"2026-07-05T10:41:59.663647+00:00"},{"alias_kind":"pith_short_12","alias_value":"NNNS6ZVMF3BO","created_at":"2026-07-05T10:41:59.663647+00:00"},{"alias_kind":"pith_short_16","alias_value":"NNNS6ZVMF3BOZKAN","created_at":"2026-07-05T10:41:59.663647+00:00"},{"alias_kind":"pith_short_8","alias_value":"NNNS6ZVM","created_at":"2026-07-05T10:41:59.663647+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11740","citing_title":"UniReason-Med: A Shared Grounded Reasoning Interface for 2D-to-3D Transfer in Medical VQA","ref_index":73,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20623","citing_title":"RSRCC: A Remote Sensing Regional Change Comprehension Benchmark Constructed via Retrieval-Augmented Best-of-N Ranking","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NNNS6ZVMF3BOZKANOWBOCYFCVA","json":"https://pith.science/pith/NNNS6ZVMF3BOZKANOWBOCYFCVA.json","graph_json":"https://pith.science/api/pith-number/NNNS6ZVMF3BOZKANOWBOCYFCVA/graph.json","events_json":"https://pith.science/api/pith-number/NNNS6ZVMF3BOZKANOWBOCYFCVA/events.json","paper":"https://pith.science/paper/NNNS6ZVM"},"agent_actions":{"view_html":"https://pith.science/pith/NNNS6ZVMF3BOZKANOWBOCYFCVA","download_json":"https://pith.science/pith/NNNS6ZVMF3BOZKANOWBOCYFCVA.json","view_paper":"https://pith.science/paper/NNNS6ZVM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.23771&json=true","fetch_graph":"https://pith.science/api/pith-number/NNNS6ZVMF3BOZKANOWBOCYFCVA/graph.json","fetch_events":"https://pith.science/api/pith-number/NNNS6ZVMF3BOZKANOWBOCYFCVA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NNNS6ZVMF3BOZKANOWBOCYFCVA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NNNS6ZVMF3BOZKANOWBOCYFCVA/action/storage_attestation","attest_author":"https://pith.science/pith/NNNS6ZVMF3BOZKANOWBOCYFCVA/action/author_attestation","sign_citation":"https://pith.science/pith/NNNS6ZVMF3BOZKANOWBOCYFCVA/action/citation_signature","submit_replication":"https://pith.science/pith/NNNS6ZVMF3BOZKANOWBOCYFCVA/action/replication_record"}},"created_at":"2026-07-05T10:41:59.663647+00:00","updated_at":"2026-07-05T10:41:59.663647+00:00"}