{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:QMKSAIZ4YMQJRDLDPV3EJND2RI","short_pith_number":"pith:QMKSAIZ4","schema_version":"1.0","canonical_sha256":"831520233cc320988d637d7644b47a8a02a104b0e23a2ddd6aca2735a7d71a21","source":{"kind":"arxiv","id":"2406.12384","version":2},"attestation_state":"computed","paper":{"title":"VRSBench: A Versatile Vision-Language Benchmark Dataset for Remote Sensing Image Understanding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jian Ding, Mohamed Elhoseiny, Xiang Li","submitted_at":"2024-06-18T08:15:21Z","abstract_excerpt":"We introduce a new benchmark designed to advance the development of general-purpose, large-scale vision-language models for remote sensing images. Although several vision-language datasets in remote sensing have been proposed to pursue this goal, existing datasets are typically tailored to single tasks, lack detailed object information, or suffer from inadequate quality control. Exploring these improvement opportunities, we present a Versatile vision-language Benchmark for Remote Sensing image understanding, termed VRSBench. This benchmark comprises 29,614 images, with 29,614 human-verified de"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.12384","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-06-18T08:15:21Z","cross_cats_sorted":[],"title_canon_sha256":"7b60fe27bb6bcc2ed6e6b36fd1bfb8225350c045b2b67863344978edef64361a","abstract_canon_sha256":"ae1abb16bf269ad147cc0c1be4f7f1919d1b7b077bef1a062ce4d52018065286"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:33:33.693396Z","signature_b64":"MWoorUi5nWJbJ7qJgausHeHWIfP1aiF/LwygVfQFuS+2kJ3ph2lhAOneLgdvSEHIyZyCxCCJ8ANip6N+d96LCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"831520233cc320988d637d7644b47a8a02a104b0e23a2ddd6aca2735a7d71a21","last_reissued_at":"2026-07-05T09:33:33.692947Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:33:33.692947Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VRSBench: A Versatile Vision-Language Benchmark Dataset for Remote Sensing Image Understanding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jian Ding, Mohamed Elhoseiny, Xiang Li","submitted_at":"2024-06-18T08:15:21Z","abstract_excerpt":"We introduce a new benchmark designed to advance the development of general-purpose, large-scale vision-language models for remote sensing images. Although several vision-language datasets in remote sensing have been proposed to pursue this goal, existing datasets are typically tailored to single tasks, lack detailed object information, or suffer from inadequate quality control. Exploring these improvement opportunities, we present a Versatile vision-language Benchmark for Remote Sensing image understanding, termed VRSBench. This benchmark comprises 29,614 images, with 29,614 human-verified de"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.12384","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.12384/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.12384","created_at":"2026-07-05T09:33:33.692999+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.12384v2","created_at":"2026-07-05T09:33:33.692999+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.12384","created_at":"2026-07-05T09:33:33.692999+00:00"},{"alias_kind":"pith_short_12","alias_value":"QMKSAIZ4YMQJ","created_at":"2026-07-05T09:33:33.692999+00:00"},{"alias_kind":"pith_short_16","alias_value":"QMKSAIZ4YMQJRDLD","created_at":"2026-07-05T09:33:33.692999+00:00"},{"alias_kind":"pith_short_8","alias_value":"QMKSAIZ4","created_at":"2026-07-05T09:33:33.692999+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07758","citing_title":"Scalable and Trustworthy Earth Observation Foundation Models","ref_index":19,"is_internal_anchor":true},{"citing_arxiv_id":"2606.11740","citing_title":"UniReason-Med: A Shared Grounded Reasoning Interface for 2D-to-3D Transfer in Medical VQA","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2509.16343","citing_title":"Visual Reasoning Agent: Robust Vision Systems in Remote Sensing via Inference-Time Scaling","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20623","citing_title":"RSRCC: A Remote Sensing Regional Change Comprehension Benchmark Constructed via Retrieval-Augmented Best-of-N Ranking","ref_index":20,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QMKSAIZ4YMQJRDLDPV3EJND2RI","json":"https://pith.science/pith/QMKSAIZ4YMQJRDLDPV3EJND2RI.json","graph_json":"https://pith.science/api/pith-number/QMKSAIZ4YMQJRDLDPV3EJND2RI/graph.json","events_json":"https://pith.science/api/pith-number/QMKSAIZ4YMQJRDLDPV3EJND2RI/events.json","paper":"https://pith.science/paper/QMKSAIZ4"},"agent_actions":{"view_html":"https://pith.science/pith/QMKSAIZ4YMQJRDLDPV3EJND2RI","download_json":"https://pith.science/pith/QMKSAIZ4YMQJRDLDPV3EJND2RI.json","view_paper":"https://pith.science/paper/QMKSAIZ4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.12384&json=true","fetch_graph":"https://pith.science/api/pith-number/QMKSAIZ4YMQJRDLDPV3EJND2RI/graph.json","fetch_events":"https://pith.science/api/pith-number/QMKSAIZ4YMQJRDLDPV3EJND2RI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QMKSAIZ4YMQJRDLDPV3EJND2RI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QMKSAIZ4YMQJRDLDPV3EJND2RI/action/storage_attestation","attest_author":"https://pith.science/pith/QMKSAIZ4YMQJRDLDPV3EJND2RI/action/author_attestation","sign_citation":"https://pith.science/pith/QMKSAIZ4YMQJRDLDPV3EJND2RI/action/citation_signature","submit_replication":"https://pith.science/pith/QMKSAIZ4YMQJRDLDPV3EJND2RI/action/replication_record"}},"created_at":"2026-07-05T09:33:33.692999+00:00","updated_at":"2026-07-05T09:33:33.692999+00:00"}