{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:RIL5RORUGFMXPMV5ARNYY4ICHL","short_pith_number":"pith:RIL5RORU","schema_version":"1.0","canonical_sha256":"8a17d8ba34315977b2bd045b8c71023ad8b04f9766a118491a09a41df372b9bf","source":{"kind":"arxiv","id":"2410.19552","version":2},"attestation_state":"computed","paper":{"title":"GeoLLaVA: Efficient Fine-Tuned Vision-Language Models for Temporal Change Detection in Remote Sensing","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Ahmed Aboeitta, Ahmed Sharshar, Hosam Elgendy, Mohsen Guizani, Yasser Ashraf","submitted_at":"2024-10-25T13:26:32Z","abstract_excerpt":"Detecting temporal changes in geographical landscapes is critical for applications like environmental monitoring and urban planning. While remote sensing data is abundant, existing vision-language models (VLMs) often fail to capture temporal dynamics effectively. This paper addresses these limitations by introducing an annotated dataset of video frame pairs to track evolving geographical patterns over time. Using fine-tuning techniques like Low-Rank Adaptation (LoRA), quantized LoRA (QLoRA), and model pruning on models such as Video-LLaVA and LLaVA-NeXT-Video, we significantly enhance VLM perf"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.19552","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-10-25T13:26:32Z","cross_cats_sorted":[],"title_canon_sha256":"94d9a005a96257fc3079552a037d643bf3d69c1aaec340e8d7ae4ddd215d853e","abstract_canon_sha256":"fb6dd2d308b82f8b4f3f075d540eefe5adc81e3f8132ad085b1dadb9c4fddac4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:07:15.987421Z","signature_b64":"kCi5ZpWuVKqL9RDD3zwWIMeQvIH/Xro+SfUGvZ68JLPuBADP9/f2jwQPxpolgYSME6d4/J70CsyUhb34+iwNCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8a17d8ba34315977b2bd045b8c71023ad8b04f9766a118491a09a41df372b9bf","last_reissued_at":"2026-07-05T11:07:15.986908Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:07:15.986908Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GeoLLaVA: Efficient Fine-Tuned Vision-Language Models for Temporal Change Detection in Remote Sensing","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Ahmed Aboeitta, Ahmed Sharshar, Hosam Elgendy, Mohsen Guizani, Yasser Ashraf","submitted_at":"2024-10-25T13:26:32Z","abstract_excerpt":"Detecting temporal changes in geographical landscapes is critical for applications like environmental monitoring and urban planning. While remote sensing data is abundant, existing vision-language models (VLMs) often fail to capture temporal dynamics effectively. This paper addresses these limitations by introducing an annotated dataset of video frame pairs to track evolving geographical patterns over time. Using fine-tuning techniques like Low-Rank Adaptation (LoRA), quantized LoRA (QLoRA), and model pruning on models such as Video-LLaVA and LLaVA-NeXT-Video, we significantly enhance VLM perf"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.19552","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.19552/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.19552","created_at":"2026-07-05T11:07:15.986967+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.19552v2","created_at":"2026-07-05T11:07:15.986967+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.19552","created_at":"2026-07-05T11:07:15.986967+00:00"},{"alias_kind":"pith_short_12","alias_value":"RIL5RORUGFMX","created_at":"2026-07-05T11:07:15.986967+00:00"},{"alias_kind":"pith_short_16","alias_value":"RIL5RORUGFMXPMV5","created_at":"2026-07-05T11:07:15.986967+00:00"},{"alias_kind":"pith_short_8","alias_value":"RIL5RORU","created_at":"2026-07-05T11:07:15.986967+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.10819","citing_title":"Earth-OneVision: Extending Remote Sensing Multimodal Large Language Models to More Sensor Modalities and Tasks","ref_index":75,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15024","citing_title":"HiSem: Hierarchical Semantic Disentangling for Remote Sensing Image Change Captioning","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2508.10635","citing_title":"ChatENV: An Interactive Vision-Language Model for Sensor-Guided Environmental Monitoring and Scenario Simulation","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12772","citing_title":"A Multi-Agent Feedback System for Detecting and Describing News Events in Satellite Imagery","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08884","citing_title":"HM-Bench: A Comprehensive Benchmark for Multimodal Large Language Models in Hyperspectral Remote Sensing","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RIL5RORUGFMXPMV5ARNYY4ICHL","json":"https://pith.science/pith/RIL5RORUGFMXPMV5ARNYY4ICHL.json","graph_json":"https://pith.science/api/pith-number/RIL5RORUGFMXPMV5ARNYY4ICHL/graph.json","events_json":"https://pith.science/api/pith-number/RIL5RORUGFMXPMV5ARNYY4ICHL/events.json","paper":"https://pith.science/paper/RIL5RORU"},"agent_actions":{"view_html":"https://pith.science/pith/RIL5RORUGFMXPMV5ARNYY4ICHL","download_json":"https://pith.science/pith/RIL5RORUGFMXPMV5ARNYY4ICHL.json","view_paper":"https://pith.science/paper/RIL5RORU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.19552&json=true","fetch_graph":"https://pith.science/api/pith-number/RIL5RORUGFMXPMV5ARNYY4ICHL/graph.json","fetch_events":"https://pith.science/api/pith-number/RIL5RORUGFMXPMV5ARNYY4ICHL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RIL5RORUGFMXPMV5ARNYY4ICHL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RIL5RORUGFMXPMV5ARNYY4ICHL/action/storage_attestation","attest_author":"https://pith.science/pith/RIL5RORUGFMXPMV5ARNYY4ICHL/action/author_attestation","sign_citation":"https://pith.science/pith/RIL5RORUGFMXPMV5ARNYY4ICHL/action/citation_signature","submit_replication":"https://pith.science/pith/RIL5RORUGFMXPMV5ARNYY4ICHL/action/replication_record"}},"created_at":"2026-07-05T11:07:15.986967+00:00","updated_at":"2026-07-05T11:07:15.986967+00:00"}