{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:4ZXQHJ5C4ZQWBBBSISZF3OA54P","short_pith_number":"pith:4ZXQHJ5C","schema_version":"1.0","canonical_sha256":"e66f03a7a2e66160843244b25db81de3dd886438f142593efc7fffd1c6a710ce","source":{"kind":"arxiv","id":"2508.07918","version":1},"attestation_state":"computed","paper":{"title":"RSVLM-QA: A Benchmark Dataset for Remote Sensing Vision Language Model-based Question Answering","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Ali Braytee, Jinghao Xiao, Jun Li, Mukesh Prasad, Xian Tao, Xing Zi, Yunxiao Shi","submitted_at":"2025-08-11T12:32:48Z","abstract_excerpt":"Visual Question Answering (VQA) in remote sensing (RS) is pivotal for interpreting Earth observation data. However, existing RS VQA datasets are constrained by limitations in annotation richness, question diversity, and the assessment of specific reasoning capabilities. This paper introduces RSVLM-QA dataset, a new large-scale, content-rich VQA dataset for the RS domain. RSVLM-QA is constructed by integrating data from several prominent RS segmentation and detection datasets: WHU, LoveDA, INRIA, and iSAID. We employ an innovative dual-track annotation generation pipeline. Firstly, we leverage "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.07918","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-08-11T12:32:48Z","cross_cats_sorted":[],"title_canon_sha256":"59162a95db5cbcb99c6a049fee95e7fe1fccf575625420b1eb2d0a4afb631d9c","abstract_canon_sha256":"aa2f7c216921b3aebba60602f6cf149d00524eeaceab1b72c47443be825c1367"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:51:54.486478Z","signature_b64":"NB5+NtskoWcfDyR0tRP/sYhfv90Hwbp+H1kad3DBQ2D5FXLLrfIL0P7GE21xZB2WUtgDLTuqCQoBOaH203E+Ag==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e66f03a7a2e66160843244b25db81de3dd886438f142593efc7fffd1c6a710ce","last_reissued_at":"2026-07-05T11:51:54.485735Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:51:54.485735Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RSVLM-QA: A Benchmark Dataset for Remote Sensing Vision Language Model-based Question Answering","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Ali Braytee, Jinghao Xiao, Jun Li, Mukesh Prasad, Xian Tao, Xing Zi, Yunxiao Shi","submitted_at":"2025-08-11T12:32:48Z","abstract_excerpt":"Visual Question Answering (VQA) in remote sensing (RS) is pivotal for interpreting Earth observation data. However, existing RS VQA datasets are constrained by limitations in annotation richness, question diversity, and the assessment of specific reasoning capabilities. This paper introduces RSVLM-QA dataset, a new large-scale, content-rich VQA dataset for the RS domain. RSVLM-QA is constructed by integrating data from several prominent RS segmentation and detection datasets: WHU, LoveDA, INRIA, and iSAID. We employ an innovative dual-track annotation generation pipeline. Firstly, we leverage "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.07918","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.07918/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.07918","created_at":"2026-07-05T11:51:54.486000+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.07918v1","created_at":"2026-07-05T11:51:54.486000+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.07918","created_at":"2026-07-05T11:51:54.486000+00:00"},{"alias_kind":"pith_short_12","alias_value":"4ZXQHJ5C4ZQW","created_at":"2026-07-05T11:51:54.486000+00:00"},{"alias_kind":"pith_short_16","alias_value":"4ZXQHJ5C4ZQWBBBS","created_at":"2026-07-05T11:51:54.486000+00:00"},{"alias_kind":"pith_short_8","alias_value":"4ZXQHJ5C","created_at":"2026-07-05T11:51:54.486000+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03054","citing_title":"ToolGate: Token-Efficient Pre-Call Control for Tool-Augmented Vision-Language Agents","ref_index":67,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4ZXQHJ5C4ZQWBBBSISZF3OA54P","json":"https://pith.science/pith/4ZXQHJ5C4ZQWBBBSISZF3OA54P.json","graph_json":"https://pith.science/api/pith-number/4ZXQHJ5C4ZQWBBBSISZF3OA54P/graph.json","events_json":"https://pith.science/api/pith-number/4ZXQHJ5C4ZQWBBBSISZF3OA54P/events.json","paper":"https://pith.science/paper/4ZXQHJ5C"},"agent_actions":{"view_html":"https://pith.science/pith/4ZXQHJ5C4ZQWBBBSISZF3OA54P","download_json":"https://pith.science/pith/4ZXQHJ5C4ZQWBBBSISZF3OA54P.json","view_paper":"https://pith.science/paper/4ZXQHJ5C","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.07918&json=true","fetch_graph":"https://pith.science/api/pith-number/4ZXQHJ5C4ZQWBBBSISZF3OA54P/graph.json","fetch_events":"https://pith.science/api/pith-number/4ZXQHJ5C4ZQWBBBSISZF3OA54P/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4ZXQHJ5C4ZQWBBBSISZF3OA54P/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4ZXQHJ5C4ZQWBBBSISZF3OA54P/action/storage_attestation","attest_author":"https://pith.science/pith/4ZXQHJ5C4ZQWBBBSISZF3OA54P/action/author_attestation","sign_citation":"https://pith.science/pith/4ZXQHJ5C4ZQWBBBSISZF3OA54P/action/citation_signature","submit_replication":"https://pith.science/pith/4ZXQHJ5C4ZQWBBBSISZF3OA54P/action/replication_record"}},"created_at":"2026-07-05T11:51:54.486000+00:00","updated_at":"2026-07-05T11:51:54.486000+00:00"}