{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:B4LCQ3OVYRECUUKBOHIFVRL4BT","short_pith_number":"pith:B4LCQ3OV","schema_version":"1.0","canonical_sha256":"0f16286dd5c4482a514171d05ac57c0cd7735059fbc64277c7b99415bd96a03f","source":{"kind":"arxiv","id":"2507.16716","version":1},"attestation_state":"computed","paper":{"title":"Enhancing Remote Sensing Vision-Language Models Through MLLM and LLM-Based High-Quality Image-Text Dataset Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chunping Qiu, Junjie Zhu, Jun Wang, Ke Yang, Qiangjuan Huang, Xiaoyu Zhang, Yiguo He, Yiying Li","submitted_at":"2025-07-22T15:54:53Z","abstract_excerpt":"The application of Vision-language foundation models (VLFMs) to remote sensing (RS) imagery has garnered significant attention due to their superior capability in various downstream tasks. A key challenge lies in the scarcity of high-quality, large-scale, image-text paired training data. Recently, several works introduced extensive image-text datasets for RS and trained their VLFMs. However, due to the rudimentary methods used for generating captions, the quality of datasets is suboptimal, requiring larger volumes of training data, while only yielding modest performance improvements. In this p"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.16716","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-07-22T15:54:53Z","cross_cats_sorted":[],"title_canon_sha256":"8bb98f2eec236b6b6bc9e1956dd1b90e6676bd8d17d5e7817256e87b6e5aaee0","abstract_canon_sha256":"c4c733f842bf3f8e96c2bdc566b6140f7275379b0a904f3ed55d2be5eaae1537"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:41:29.776147Z","signature_b64":"gPpH0ohF0ayj8WG/h/W65wg30lSZLMKnsIpscBV3JuGiLfzL+RmcRtknfLtS06UAVY6Vvrxs8rzC0PIpWCRiCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0f16286dd5c4482a514171d05ac57c0cd7735059fbc64277c7b99415bd96a03f","last_reissued_at":"2026-07-05T11:41:29.775636Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:41:29.775636Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Enhancing Remote Sensing Vision-Language Models Through MLLM and LLM-Based High-Quality Image-Text Dataset Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chunping Qiu, Junjie Zhu, Jun Wang, Ke Yang, Qiangjuan Huang, Xiaoyu Zhang, Yiguo He, Yiying Li","submitted_at":"2025-07-22T15:54:53Z","abstract_excerpt":"The application of Vision-language foundation models (VLFMs) to remote sensing (RS) imagery has garnered significant attention due to their superior capability in various downstream tasks. A key challenge lies in the scarcity of high-quality, large-scale, image-text paired training data. Recently, several works introduced extensive image-text datasets for RS and trained their VLFMs. However, due to the rudimentary methods used for generating captions, the quality of datasets is suboptimal, requiring larger volumes of training data, while only yielding modest performance improvements. In this p"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.16716","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.16716/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.16716","created_at":"2026-07-05T11:41:29.775696+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.16716v1","created_at":"2026-07-05T11:41:29.775696+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.16716","created_at":"2026-07-05T11:41:29.775696+00:00"},{"alias_kind":"pith_short_12","alias_value":"B4LCQ3OVYREC","created_at":"2026-07-05T11:41:29.775696+00:00"},{"alias_kind":"pith_short_16","alias_value":"B4LCQ3OVYRECUUKB","created_at":"2026-07-05T11:41:29.775696+00:00"},{"alias_kind":"pith_short_8","alias_value":"B4LCQ3OV","created_at":"2026-07-05T11:41:29.775696+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.10819","citing_title":"Earth-OneVision: Extending Remote Sensing Multimodal Large Language Models to More Sensor Modalities and Tasks","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2510.22665","citing_title":"SARVLM: A Vision Language Foundation Model for Semantic Understanding in SAR Imagery","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15558","citing_title":"Text-RSIR: A Text-Guided Framework for Efficient Remote Sensing Image Transmission and Reconstruction","ref_index":58,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/B4LCQ3OVYRECUUKBOHIFVRL4BT","json":"https://pith.science/pith/B4LCQ3OVYRECUUKBOHIFVRL4BT.json","graph_json":"https://pith.science/api/pith-number/B4LCQ3OVYRECUUKBOHIFVRL4BT/graph.json","events_json":"https://pith.science/api/pith-number/B4LCQ3OVYRECUUKBOHIFVRL4BT/events.json","paper":"https://pith.science/paper/B4LCQ3OV"},"agent_actions":{"view_html":"https://pith.science/pith/B4LCQ3OVYRECUUKBOHIFVRL4BT","download_json":"https://pith.science/pith/B4LCQ3OVYRECUUKBOHIFVRL4BT.json","view_paper":"https://pith.science/paper/B4LCQ3OV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.16716&json=true","fetch_graph":"https://pith.science/api/pith-number/B4LCQ3OVYRECUUKBOHIFVRL4BT/graph.json","fetch_events":"https://pith.science/api/pith-number/B4LCQ3OVYRECUUKBOHIFVRL4BT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/B4LCQ3OVYRECUUKBOHIFVRL4BT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/B4LCQ3OVYRECUUKBOHIFVRL4BT/action/storage_attestation","attest_author":"https://pith.science/pith/B4LCQ3OVYRECUUKBOHIFVRL4BT/action/author_attestation","sign_citation":"https://pith.science/pith/B4LCQ3OVYRECUUKBOHIFVRL4BT/action/citation_signature","submit_replication":"https://pith.science/pith/B4LCQ3OVYRECUUKBOHIFVRL4BT/action/replication_record"}},"created_at":"2026-07-05T11:41:29.775696+00:00","updated_at":"2026-07-05T11:41:29.775696+00:00"}