{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:KJW7PJBLKTAEP5BQHXJFZOVGEN","short_pith_number":"pith:KJW7PJBL","schema_version":"1.0","canonical_sha256":"526df7a42b54c047f4303dd25cbaa62348743756619919b1a07e5673fed43bf8","source":{"kind":"arxiv","id":"2410.06234","version":2},"attestation_state":"computed","paper":{"title":"TEOChat: A Large Vision-Language Assistant for Temporal Earth Observation Data","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Emily Ruoyu Liu, Ines Dormoy, Jeremy Andrew Irvin, Jinyoung Kim, Joyce Chuyi Chen, Samar Khanna, Stefano Ermon, Zhuo Zheng","submitted_at":"2024-10-08T17:45:51Z","abstract_excerpt":"Large vision and language assistants have enabled new capabilities for interpreting natural images. These approaches have recently been adapted to earth observation data, but they are only able to handle single image inputs, limiting their use for many real-world tasks. In this work, we develop a new vision and language assistant called TEOChat that can engage in conversations about temporal sequences of earth observation data. To train TEOChat, we curate an instruction-following dataset composed of many single image and temporal tasks including building change and damage assessment, semantic "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.06234","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-10-08T17:45:51Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"2bf3ecc0882d29bd3adaf69ded6166fdc8cb7db8d56229c14d34ebb40cbaaf6f","abstract_canon_sha256":"4838843e75b515cfed520b7942400f9fd20d419c6530aed0d84c4b0e3653e889"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:05:45.517499Z","signature_b64":"FkLCpfXXToSpGDiGaehs0VESjrNylfWRlnRoY9mvFNe+e3ghrkNQFDv9kb7NbaEB3w6ZqFFxj040DK6JOzf4Bw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"526df7a42b54c047f4303dd25cbaa62348743756619919b1a07e5673fed43bf8","last_reissued_at":"2026-07-05T10:05:45.516884Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:05:45.516884Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TEOChat: A Large Vision-Language Assistant for Temporal Earth Observation Data","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Emily Ruoyu Liu, Ines Dormoy, Jeremy Andrew Irvin, Jinyoung Kim, Joyce Chuyi Chen, Samar Khanna, Stefano Ermon, Zhuo Zheng","submitted_at":"2024-10-08T17:45:51Z","abstract_excerpt":"Large vision and language assistants have enabled new capabilities for interpreting natural images. These approaches have recently been adapted to earth observation data, but they are only able to handle single image inputs, limiting their use for many real-world tasks. In this work, we develop a new vision and language assistant called TEOChat that can engage in conversations about temporal sequences of earth observation data. To train TEOChat, we curate an instruction-following dataset composed of many single image and temporal tasks including building change and damage assessment, semantic "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.06234","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.06234/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.06234","created_at":"2026-07-05T10:05:45.516956+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.06234v2","created_at":"2026-07-05T10:05:45.516956+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.06234","created_at":"2026-07-05T10:05:45.516956+00:00"},{"alias_kind":"pith_short_12","alias_value":"KJW7PJBLKTAE","created_at":"2026-07-05T10:05:45.516956+00:00"},{"alias_kind":"pith_short_16","alias_value":"KJW7PJBLKTAEP5BQ","created_at":"2026-07-05T10:05:45.516956+00:00"},{"alias_kind":"pith_short_8","alias_value":"KJW7PJBL","created_at":"2026-07-05T10:05:45.516956+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11740","citing_title":"UniReason-Med: A Shared Grounded Reasoning Interface for 2D-to-3D Transfer in Medical VQA","ref_index":141,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28266","citing_title":"RSICCLLM: A Multimodal Large Language Model for Remote Sensing Image Change Captioning","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15024","citing_title":"HiSem: Hierarchical Semantic Disentangling for Remote Sensing Image Change Captioning","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12542","citing_title":"Earth Science Foundation Models: From Perception to Reasoning and Discovery","ref_index":66,"is_internal_anchor":false},{"citing_arxiv_id":"2508.10635","citing_title":"ChatENV: An Interactive Vision-Language Model for Sensor-Guided Environmental Monitoring and Scenario Simulation","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12542","citing_title":"Earth Science Foundation Models: From Perception to Reasoning and Discovery","ref_index":66,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10576","citing_title":"SenseBench: A Benchmark for Remote Sensing Low-Level Visual Perception and Description in Large Vision-Language Models","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12772","citing_title":"A Multi-Agent Feedback System for Detecting and Describing News Events in Satellite Imagery","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14044","citing_title":"Decoding the Delta: Unifying Remote Sensing Change Detection and Understanding with Multimodal Large Language Models","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17243","citing_title":"RemoteShield: Enable Robust Multimodal Large Language Models for Earth Observation","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KJW7PJBLKTAEP5BQHXJFZOVGEN","json":"https://pith.science/pith/KJW7PJBLKTAEP5BQHXJFZOVGEN.json","graph_json":"https://pith.science/api/pith-number/KJW7PJBLKTAEP5BQHXJFZOVGEN/graph.json","events_json":"https://pith.science/api/pith-number/KJW7PJBLKTAEP5BQHXJFZOVGEN/events.json","paper":"https://pith.science/paper/KJW7PJBL"},"agent_actions":{"view_html":"https://pith.science/pith/KJW7PJBLKTAEP5BQHXJFZOVGEN","download_json":"https://pith.science/pith/KJW7PJBLKTAEP5BQHXJFZOVGEN.json","view_paper":"https://pith.science/paper/KJW7PJBL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.06234&json=true","fetch_graph":"https://pith.science/api/pith-number/KJW7PJBLKTAEP5BQHXJFZOVGEN/graph.json","fetch_events":"https://pith.science/api/pith-number/KJW7PJBLKTAEP5BQHXJFZOVGEN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KJW7PJBLKTAEP5BQHXJFZOVGEN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KJW7PJBLKTAEP5BQHXJFZOVGEN/action/storage_attestation","attest_author":"https://pith.science/pith/KJW7PJBLKTAEP5BQHXJFZOVGEN/action/author_attestation","sign_citation":"https://pith.science/pith/KJW7PJBLKTAEP5BQHXJFZOVGEN/action/citation_signature","submit_replication":"https://pith.science/pith/KJW7PJBLKTAEP5BQHXJFZOVGEN/action/replication_record"}},"created_at":"2026-07-05T10:05:45.516956+00:00","updated_at":"2026-07-05T10:05:45.516956+00:00"}