{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:Q3CVGRLKR7RA7L6A525EVUMJKD","short_pith_number":"pith:Q3CVGRLK","schema_version":"1.0","canonical_sha256":"86c553456a8fe20fafc0eeba4ad18950e6ab5ab97145b853116676f4d4c47fd1","source":{"kind":"arxiv","id":"2506.21863","version":1},"attestation_state":"computed","paper":{"title":"Remote Sensing Large Vision-Language Model: Semantic-augmented Multi-level Alignment and Semantic-aware Expert Modeling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Se Yeon Kim, Sungjune Park, Yeongyun Kim, Yong Man Ro","submitted_at":"2025-06-27T02:31:37Z","abstract_excerpt":"Large Vision and Language Models (LVLMs) have shown strong performance across various vision-language tasks in natural image domains. However, their application to remote sensing (RS) remains underexplored due to significant domain differences in visual appearances, object scales, and semantics. These discrepancies hider the effective understanding of RS scenes, which contain rich, multi-level semantic information spanning from coarse-to-fine levels. Hence, it limits the direct adaptation of existing LVLMs to RS imagery. To address this gap, we propose a novel LVLM framework tailored for RS un"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.21863","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-06-27T02:31:37Z","cross_cats_sorted":[],"title_canon_sha256":"d215283af2ce659cb2f6079da8d7fd7f2c3f8d5c9dfa0222dd9c5eeae3779436","abstract_canon_sha256":"6e663d908aa7c7f8eacaaaae9326c5cc6f35a22cde9869f01d807f148bf4d5df"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:28:02.959341Z","signature_b64":"B4V02udsODjhsIQIsjQr7O3HdVByFiFGJ/yTW8Rv5y6cpultJlNqqMVWyYAkJwhssb/6Y2eqEHSp01fvlD+NBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"86c553456a8fe20fafc0eeba4ad18950e6ab5ab97145b853116676f4d4c47fd1","last_reissued_at":"2026-07-05T11:28:02.958843Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:28:02.958843Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Remote Sensing Large Vision-Language Model: Semantic-augmented Multi-level Alignment and Semantic-aware Expert Modeling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Se Yeon Kim, Sungjune Park, Yeongyun Kim, Yong Man Ro","submitted_at":"2025-06-27T02:31:37Z","abstract_excerpt":"Large Vision and Language Models (LVLMs) have shown strong performance across various vision-language tasks in natural image domains. However, their application to remote sensing (RS) remains underexplored due to significant domain differences in visual appearances, object scales, and semantics. These discrepancies hider the effective understanding of RS scenes, which contain rich, multi-level semantic information spanning from coarse-to-fine levels. Hence, it limits the direct adaptation of existing LVLMs to RS imagery. To address this gap, we propose a novel LVLM framework tailored for RS un"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.21863","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.21863/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.21863","created_at":"2026-07-05T11:28:02.958911+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.21863v1","created_at":"2026-07-05T11:28:02.958911+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.21863","created_at":"2026-07-05T11:28:02.958911+00:00"},{"alias_kind":"pith_short_12","alias_value":"Q3CVGRLKR7RA","created_at":"2026-07-05T11:28:02.958911+00:00"},{"alias_kind":"pith_short_16","alias_value":"Q3CVGRLKR7RA7L6A","created_at":"2026-07-05T11:28:02.958911+00:00"},{"alias_kind":"pith_short_8","alias_value":"Q3CVGRLK","created_at":"2026-07-05T11:28:02.958911+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.13654","citing_title":"Vision-and-Language Navigation for UAVs: Progress, Challenges, and a Research Roadmap","ref_index":148,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Q3CVGRLKR7RA7L6A525EVUMJKD","json":"https://pith.science/pith/Q3CVGRLKR7RA7L6A525EVUMJKD.json","graph_json":"https://pith.science/api/pith-number/Q3CVGRLKR7RA7L6A525EVUMJKD/graph.json","events_json":"https://pith.science/api/pith-number/Q3CVGRLKR7RA7L6A525EVUMJKD/events.json","paper":"https://pith.science/paper/Q3CVGRLK"},"agent_actions":{"view_html":"https://pith.science/pith/Q3CVGRLKR7RA7L6A525EVUMJKD","download_json":"https://pith.science/pith/Q3CVGRLKR7RA7L6A525EVUMJKD.json","view_paper":"https://pith.science/paper/Q3CVGRLK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.21863&json=true","fetch_graph":"https://pith.science/api/pith-number/Q3CVGRLKR7RA7L6A525EVUMJKD/graph.json","fetch_events":"https://pith.science/api/pith-number/Q3CVGRLKR7RA7L6A525EVUMJKD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Q3CVGRLKR7RA7L6A525EVUMJKD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Q3CVGRLKR7RA7L6A525EVUMJKD/action/storage_attestation","attest_author":"https://pith.science/pith/Q3CVGRLKR7RA7L6A525EVUMJKD/action/author_attestation","sign_citation":"https://pith.science/pith/Q3CVGRLKR7RA7L6A525EVUMJKD/action/citation_signature","submit_replication":"https://pith.science/pith/Q3CVGRLKR7RA7L6A525EVUMJKD/action/replication_record"}},"created_at":"2026-07-05T11:28:02.958911+00:00","updated_at":"2026-07-05T11:28:02.958911+00:00"}