{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:6H2UN2LYD6MWOY2U44DVJYKLOX","short_pith_number":"pith:6H2UN2LY","schema_version":"1.0","canonical_sha256":"f1f546e9781f99676354e70754e14b75dd645b44f458acd8d04776240e06f6da","source":{"kind":"arxiv","id":"2406.10100","version":2},"attestation_state":"computed","paper":{"title":"SkySenseGPT: A Fine-Grained Instruction Tuning Dataset and Model for Remote Sensing Vision-Language Understanding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Bo Dang, Jiangwei Lao, Jian Wang, Jingdong Chen, Junwei Luo, Linlin Wang, Tingzhu Wang, Yansheng Li, Yihua Tan, Yongjun Zhang, Zhen Pang","submitted_at":"2024-06-14T14:57:07Z","abstract_excerpt":"Remote Sensing Large Multi-Modal Models (RSLMMs) are developing rapidly and showcase significant capabilities in remote sensing imagery (RSI) comprehension. However, due to the limitations of existing datasets, RSLMMs have shortcomings in understanding the rich semantic relations among objects in complex remote sensing scenes. To unlock RSLMMs' complex comprehension ability, we propose a large-scale instruction tuning dataset FIT-RS, containing 1,800,851 instruction samples. FIT-RS covers common interpretation tasks and innovatively introduces several complex comprehension tasks of escalating "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.10100","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-06-14T14:57:07Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"35ae0c1a0a9c058076186c2fbc222bd21996a71e78a94cc1f7dcc5173871dc66","abstract_canon_sha256":"e2f653d31d3eff136f1d554fc6480bb68a1e7c1de9744529a8eda4f13d7474e3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:41:04.033681Z","signature_b64":"S1cdxOxwHjP9Tsl2PB4vWWuzFYSc9WPlOwC3D2LVxxk3KsL8QvqERlzZrBy21HsG2OtNRSBDb8BZ09+CDz7GCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f1f546e9781f99676354e70754e14b75dd645b44f458acd8d04776240e06f6da","last_reissued_at":"2026-07-05T08:41:04.033180Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:41:04.033180Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SkySenseGPT: A Fine-Grained Instruction Tuning Dataset and Model for Remote Sensing Vision-Language Understanding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Bo Dang, Jiangwei Lao, Jian Wang, Jingdong Chen, Junwei Luo, Linlin Wang, Tingzhu Wang, Yansheng Li, Yihua Tan, Yongjun Zhang, Zhen Pang","submitted_at":"2024-06-14T14:57:07Z","abstract_excerpt":"Remote Sensing Large Multi-Modal Models (RSLMMs) are developing rapidly and showcase significant capabilities in remote sensing imagery (RSI) comprehension. However, due to the limitations of existing datasets, RSLMMs have shortcomings in understanding the rich semantic relations among objects in complex remote sensing scenes. To unlock RSLMMs' complex comprehension ability, we propose a large-scale instruction tuning dataset FIT-RS, containing 1,800,851 instruction samples. FIT-RS covers common interpretation tasks and innovatively introduces several complex comprehension tasks of escalating "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.10100","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.10100/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.10100","created_at":"2026-07-05T08:41:04.033244+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.10100v2","created_at":"2026-07-05T08:41:04.033244+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.10100","created_at":"2026-07-05T08:41:04.033244+00:00"},{"alias_kind":"pith_short_12","alias_value":"6H2UN2LYD6MW","created_at":"2026-07-05T08:41:04.033244+00:00"},{"alias_kind":"pith_short_16","alias_value":"6H2UN2LYD6MWOY2U","created_at":"2026-07-05T08:41:04.033244+00:00"},{"alias_kind":"pith_short_8","alias_value":"6H2UN2LY","created_at":"2026-07-05T08:41:04.033244+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":21,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11740","citing_title":"UniReason-Med: A Shared Grounded Reasoning Interface for 2D-to-3D Transfer in Medical VQA","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10819","citing_title":"Earth-OneVision: Extending Remote Sensing Multimodal Large Language Models to More Sensor Modalities and Tasks","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01050","citing_title":"GeoSearcher: Anchor-Guided Progressive Reasoning for Remote Sensing Visual Grounding with Process Supervision","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12542","citing_title":"Earth Science Foundation Models: From Perception to Reasoning and Discovery","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2511.23332","citing_title":"UniGeoSeg: Towards Unified Open-World Segmentation for Geospatial Scenes","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2512.17492","citing_title":"MMLANDMARKS: a Cross-View Instance-Level Benchmark for Geo-Spatial Understanding","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2601.01891","citing_title":"Agentic AI in Remote Sensing: Foundations, Taxonomy, and Emerging Systems","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2602.07045","citing_title":"VLRS-Bench: A Vision-Language Reasoning Benchmark for Remote Sensing","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12542","citing_title":"Earth Science Foundation Models: From Perception to Reasoning and Discovery","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24919","citing_title":"Agentic AI for Remote Sensing: Technical Challenges and Research Directions","ref_index":75,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24919","citing_title":"Agentic AI for Remote Sensing: Technical Challenges and Research Directions","ref_index":75,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05594","citing_title":"The Cost of Context: Mitigating Textual Bias in Multimodal Retrieval-Augmented Generation","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04777","citing_title":"Bridging Perception and Action: A Lightweight Multimodal Meta-Planner Framework for Robust Earth Observation Agents","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13654","citing_title":"Vision-and-Language Navigation for UAVs: Progress, Challenges, and a Research Roadmap","ref_index":235,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10591","citing_title":"GeoMeld: Toward Semantically Grounded Foundation Models for Remote Sensing","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08896","citing_title":"GeoMMBench and GeoMMAgent: Toward Expert-Level Multimodal Intelligence in Geoscience and Remote Sensing","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07765","citing_title":"RemoteAgent: Bridging Vague Human Intents and Earth Observation with RL-based Agentic MLLMs","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07562","citing_title":"Beyond GSD-as-Token: Continuous Scale Conditioning for Remote Sensing VLMs","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17243","citing_title":"RemoteShield: Enable Robust Multimodal Large Language Models for Earth Observation","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22855","citing_title":"Evaluating Remote Sensing Image Captions Beyond Metric Biases","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04451","citing_title":"RemoteZero: Geospatial Reasoning with Zero Labels","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6H2UN2LYD6MWOY2U44DVJYKLOX","json":"https://pith.science/pith/6H2UN2LYD6MWOY2U44DVJYKLOX.json","graph_json":"https://pith.science/api/pith-number/6H2UN2LYD6MWOY2U44DVJYKLOX/graph.json","events_json":"https://pith.science/api/pith-number/6H2UN2LYD6MWOY2U44DVJYKLOX/events.json","paper":"https://pith.science/paper/6H2UN2LY"},"agent_actions":{"view_html":"https://pith.science/pith/6H2UN2LYD6MWOY2U44DVJYKLOX","download_json":"https://pith.science/pith/6H2UN2LYD6MWOY2U44DVJYKLOX.json","view_paper":"https://pith.science/paper/6H2UN2LY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.10100&json=true","fetch_graph":"https://pith.science/api/pith-number/6H2UN2LYD6MWOY2U44DVJYKLOX/graph.json","fetch_events":"https://pith.science/api/pith-number/6H2UN2LYD6MWOY2U44DVJYKLOX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6H2UN2LYD6MWOY2U44DVJYKLOX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6H2UN2LYD6MWOY2U44DVJYKLOX/action/storage_attestation","attest_author":"https://pith.science/pith/6H2UN2LYD6MWOY2U44DVJYKLOX/action/author_attestation","sign_citation":"https://pith.science/pith/6H2UN2LYD6MWOY2U44DVJYKLOX/action/citation_signature","submit_replication":"https://pith.science/pith/6H2UN2LYD6MWOY2U44DVJYKLOX/action/replication_record"}},"created_at":"2026-07-05T08:41:04.033244+00:00","updated_at":"2026-07-05T08:41:04.033244+00:00"}