{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:6GYIJDEVMCHK6X6Q5IU7NH7LRO","short_pith_number":"pith:6GYIJDEV","schema_version":"1.0","canonical_sha256":"f1b0848c95608eaf5fd0ea29f69feb8b84c16563708518d5712169ee7dc7cd27","source":{"kind":"arxiv","id":"2503.12490","version":1},"attestation_state":"computed","paper":{"title":"GeoRSMLLM: A Multimodal Large Language Model for Vision-Language Tasks in Geoscience and Remote Sensing","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Bin Chen, Haozhan Shen, Jianwei Yin, Tiancheng Zhao, Xu Jia, Yongheng Shang, Yuhao Wang, Yuxiang Cai, Zian Guan, Zilun Zhang","submitted_at":"2025-03-16T12:48:17Z","abstract_excerpt":"The application of Vision-Language Models (VLMs) in remote sensing (RS) has demonstrated significant potential in traditional tasks such as scene classification, object detection, and image captioning. However, current models, which excel in Referring Expression Comprehension (REC), struggle with tasks involving complex instructions (e.g., exists multiple conditions) or pixel-level operations like segmentation and change detection. In this white paper, we provide a comprehensive hierarchical summary of vision-language tasks in RS, categorized by the varying levels of cognitive capability requi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.12490","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2025-03-16T12:48:17Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"45b119c98c405c83c0011171dcd1824e567e6fc3f49083cd67116fce1728de14","abstract_canon_sha256":"be3df329f325f00e8e2a64d625a63840ff9ebf4b9bcbea96c29d1dee77a7d4a1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:32:14.670909Z","signature_b64":"Py89TD6hRatFiU1GYPeizGuTHrmPEW9StFd45iDPq7F40FOZKRdoLGCQuAzMXyYf9BzW32GH5YrUknWb4eKHAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f1b0848c95608eaf5fd0ea29f69feb8b84c16563708518d5712169ee7dc7cd27","last_reissued_at":"2026-07-05T10:32:14.670419Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:32:14.670419Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GeoRSMLLM: A Multimodal Large Language Model for Vision-Language Tasks in Geoscience and Remote Sensing","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Bin Chen, Haozhan Shen, Jianwei Yin, Tiancheng Zhao, Xu Jia, Yongheng Shang, Yuhao Wang, Yuxiang Cai, Zian Guan, Zilun Zhang","submitted_at":"2025-03-16T12:48:17Z","abstract_excerpt":"The application of Vision-Language Models (VLMs) in remote sensing (RS) has demonstrated significant potential in traditional tasks such as scene classification, object detection, and image captioning. However, current models, which excel in Referring Expression Comprehension (REC), struggle with tasks involving complex instructions (e.g., exists multiple conditions) or pixel-level operations like segmentation and change detection. In this white paper, we provide a comprehensive hierarchical summary of vision-language tasks in RS, categorized by the varying levels of cognitive capability requi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.12490","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.12490/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.12490","created_at":"2026-07-05T10:32:14.670477+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.12490v1","created_at":"2026-07-05T10:32:14.670477+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.12490","created_at":"2026-07-05T10:32:14.670477+00:00"},{"alias_kind":"pith_short_12","alias_value":"6GYIJDEVMCHK","created_at":"2026-07-05T10:32:14.670477+00:00"},{"alias_kind":"pith_short_16","alias_value":"6GYIJDEVMCHK6X6Q","created_at":"2026-07-05T10:32:14.670477+00:00"},{"alias_kind":"pith_short_8","alias_value":"6GYIJDEV","created_at":"2026-07-05T10:32:14.670477+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2508.10635","citing_title":"ChatENV: An Interactive Vision-Language Model for Sensor-Guided Environmental Monitoring and Scenario Simulation","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2601.01891","citing_title":"Agentic AI in Remote Sensing: Foundations, Taxonomy, and Emerging Systems","ref_index":152,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14044","citing_title":"Decoding the Delta: Unifying Remote Sensing Change Detection and Understanding with Multimodal Large Language Models","ref_index":62,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6GYIJDEVMCHK6X6Q5IU7NH7LRO","json":"https://pith.science/pith/6GYIJDEVMCHK6X6Q5IU7NH7LRO.json","graph_json":"https://pith.science/api/pith-number/6GYIJDEVMCHK6X6Q5IU7NH7LRO/graph.json","events_json":"https://pith.science/api/pith-number/6GYIJDEVMCHK6X6Q5IU7NH7LRO/events.json","paper":"https://pith.science/paper/6GYIJDEV"},"agent_actions":{"view_html":"https://pith.science/pith/6GYIJDEVMCHK6X6Q5IU7NH7LRO","download_json":"https://pith.science/pith/6GYIJDEVMCHK6X6Q5IU7NH7LRO.json","view_paper":"https://pith.science/paper/6GYIJDEV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.12490&json=true","fetch_graph":"https://pith.science/api/pith-number/6GYIJDEVMCHK6X6Q5IU7NH7LRO/graph.json","fetch_events":"https://pith.science/api/pith-number/6GYIJDEVMCHK6X6Q5IU7NH7LRO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6GYIJDEVMCHK6X6Q5IU7NH7LRO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6GYIJDEVMCHK6X6Q5IU7NH7LRO/action/storage_attestation","attest_author":"https://pith.science/pith/6GYIJDEVMCHK6X6Q5IU7NH7LRO/action/author_attestation","sign_citation":"https://pith.science/pith/6GYIJDEVMCHK6X6Q5IU7NH7LRO/action/citation_signature","submit_replication":"https://pith.science/pith/6GYIJDEVMCHK6X6Q5IU7NH7LRO/action/replication_record"}},"created_at":"2026-07-05T10:32:14.670477+00:00","updated_at":"2026-07-05T10:32:14.670477+00:00"}