{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:BAYHRSLW2SWVZV34K7QRW6GVNY","short_pith_number":"pith:BAYHRSLW","schema_version":"1.0","canonical_sha256":"083078c976d4ad5cd77c57e11b78d56e3323b9bbc0fb62aa7746f04ef4cd7d8f","source":{"kind":"arxiv","id":"2505.02829","version":1},"attestation_state":"computed","paper":{"title":"LISAT: Language-Instructed Segmentation Assistant for Satellite Imagery","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"David M. Chan, Jerome Quenum, Ritwik Gupta, Trevor Darrell, Tsung-Han Wu, Wen-Han Hsieh","submitted_at":"2025-05-05T17:56:25Z","abstract_excerpt":"Segmentation models can recognize a pre-defined set of objects in images. However, models that can reason over complex user queries that implicitly refer to multiple objects of interest are still in their infancy. Recent advances in reasoning segmentation--generating segmentation masks from complex, implicit query text--demonstrate that vision-language models can operate across an open domain and produce reasonable outputs. However, our experiments show that such models struggle with complex remote-sensing imagery. In this work, we introduce LISAt, a vision-language model designed to describe "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.02829","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.AI","submitted_at":"2025-05-05T17:56:25Z","cross_cats_sorted":[],"title_canon_sha256":"fb40e11b91d7807e5032e382529de54adc2aeefed5b0bf563a36793f7033c2c5","abstract_canon_sha256":"70fb5412d4a8b06d02b4340b37158cd7775bb7fb78e81f6fa4937d7c24e4be22"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:58:43.884741Z","signature_b64":"GZQN31EwkKkG8G57l8x7vYXmu+DrK+pSPk51DHE+5ofz4ikEaYNPwXeBe01a8Q5GAAZZ0mlpNwnpyygRltBVDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"083078c976d4ad5cd77c57e11b78d56e3323b9bbc0fb62aa7746f04ef4cd7d8f","last_reissued_at":"2026-07-05T10:58:43.884233Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:58:43.884233Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LISAT: Language-Instructed Segmentation Assistant for Satellite Imagery","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"David M. Chan, Jerome Quenum, Ritwik Gupta, Trevor Darrell, Tsung-Han Wu, Wen-Han Hsieh","submitted_at":"2025-05-05T17:56:25Z","abstract_excerpt":"Segmentation models can recognize a pre-defined set of objects in images. However, models that can reason over complex user queries that implicitly refer to multiple objects of interest are still in their infancy. Recent advances in reasoning segmentation--generating segmentation masks from complex, implicit query text--demonstrate that vision-language models can operate across an open domain and produce reasonable outputs. However, our experiments show that such models struggle with complex remote-sensing imagery. In this work, we introduce LISAt, a vision-language model designed to describe "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.02829","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.02829/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.02829","created_at":"2026-07-05T10:58:43.884293+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.02829v1","created_at":"2026-07-05T10:58:43.884293+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.02829","created_at":"2026-07-05T10:58:43.884293+00:00"},{"alias_kind":"pith_short_12","alias_value":"BAYHRSLW2SWV","created_at":"2026-07-05T10:58:43.884293+00:00"},{"alias_kind":"pith_short_16","alias_value":"BAYHRSLW2SWVZV34","created_at":"2026-07-05T10:58:43.884293+00:00"},{"alias_kind":"pith_short_8","alias_value":"BAYHRSLW","created_at":"2026-07-05T10:58:43.884293+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2511.23332","citing_title":"UniGeoSeg: Towards Unified Open-World Segmentation for Geospatial Scenes","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14044","citing_title":"Decoding the Delta: Unifying Remote Sensing Change Detection and Understanding with Multimodal Large Language Models","ref_index":38,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BAYHRSLW2SWVZV34K7QRW6GVNY","json":"https://pith.science/pith/BAYHRSLW2SWVZV34K7QRW6GVNY.json","graph_json":"https://pith.science/api/pith-number/BAYHRSLW2SWVZV34K7QRW6GVNY/graph.json","events_json":"https://pith.science/api/pith-number/BAYHRSLW2SWVZV34K7QRW6GVNY/events.json","paper":"https://pith.science/paper/BAYHRSLW"},"agent_actions":{"view_html":"https://pith.science/pith/BAYHRSLW2SWVZV34K7QRW6GVNY","download_json":"https://pith.science/pith/BAYHRSLW2SWVZV34K7QRW6GVNY.json","view_paper":"https://pith.science/paper/BAYHRSLW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.02829&json=true","fetch_graph":"https://pith.science/api/pith-number/BAYHRSLW2SWVZV34K7QRW6GVNY/graph.json","fetch_events":"https://pith.science/api/pith-number/BAYHRSLW2SWVZV34K7QRW6GVNY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BAYHRSLW2SWVZV34K7QRW6GVNY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BAYHRSLW2SWVZV34K7QRW6GVNY/action/storage_attestation","attest_author":"https://pith.science/pith/BAYHRSLW2SWVZV34K7QRW6GVNY/action/author_attestation","sign_citation":"https://pith.science/pith/BAYHRSLW2SWVZV34K7QRW6GVNY/action/citation_signature","submit_replication":"https://pith.science/pith/BAYHRSLW2SWVZV34K7QRW6GVNY/action/replication_record"}},"created_at":"2026-07-05T10:58:43.884293+00:00","updated_at":"2026-07-05T10:58:43.884293+00:00"}