{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:4BXPMGPBDQ5RYRFUFTDOGHDOHD","short_pith_number":"pith:4BXPMGPB","schema_version":"1.0","canonical_sha256":"e06ef619e11c3b1c44b42cc6e31c6e38ce15c8005cfa2894cf911b5aa0ed2dc2","source":{"kind":"arxiv","id":"2410.08613","version":2},"attestation_state":"computed","paper":{"title":"Cross-Modal Bidirectional Interaction Model for Referring Remote Sensing Image Segmentation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Tianzhu Liu, Wangmeng Zuo, Yanfeng Gu, Yuzhe Sun, Zhe Dong","submitted_at":"2024-10-11T08:28:04Z","abstract_excerpt":"Given a natural language expression and a remote sensing image, the goal of referring remote sensing image segmentation (RRSIS) is to generate a pixel-level mask of the target object identified by the referring expression. In contrast to natural scenarios, expressions in RRSIS often involve complex geospatial relationships, with target objects of interest that vary significantly in scale and lack visual saliency, thereby increasing the difficulty of achieving precise segmentation. To address the aforementioned challenges, a novel RRSIS framework is proposed, termed the cross-modal bidirectiona"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.08613","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-10-11T08:28:04Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"f84fdf6b28c3ceed927502947f21be237c759f1c4d78d3c133232ffb7e35e9ba","abstract_canon_sha256":"8b7d5cdfa65dcb501f2ad44276f42231038f972082e7bc64296f8439740d1dc6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:09:01.474332Z","signature_b64":"JipbVSizV0ikv69coTh+JLUqJ6pTZhquCbB5QgdWdLRK6HHYiLPMUCcZZRE8/Bd4XPcG+VPxdo0IESWP/WCrCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e06ef619e11c3b1c44b42cc6e31c6e38ce15c8005cfa2894cf911b5aa0ed2dc2","last_reissued_at":"2026-07-05T11:09:01.473864Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:09:01.473864Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Cross-Modal Bidirectional Interaction Model for Referring Remote Sensing Image Segmentation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Tianzhu Liu, Wangmeng Zuo, Yanfeng Gu, Yuzhe Sun, Zhe Dong","submitted_at":"2024-10-11T08:28:04Z","abstract_excerpt":"Given a natural language expression and a remote sensing image, the goal of referring remote sensing image segmentation (RRSIS) is to generate a pixel-level mask of the target object identified by the referring expression. In contrast to natural scenarios, expressions in RRSIS often involve complex geospatial relationships, with target objects of interest that vary significantly in scale and lack visual saliency, thereby increasing the difficulty of achieving precise segmentation. To address the aforementioned challenges, a novel RRSIS framework is proposed, termed the cross-modal bidirectiona"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.08613","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.08613/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.08613","created_at":"2026-07-05T11:09:01.473930+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.08613v2","created_at":"2026-07-05T11:09:01.473930+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.08613","created_at":"2026-07-05T11:09:01.473930+00:00"},{"alias_kind":"pith_short_12","alias_value":"4BXPMGPBDQ5R","created_at":"2026-07-05T11:09:01.473930+00:00"},{"alias_kind":"pith_short_16","alias_value":"4BXPMGPBDQ5RYRFU","created_at":"2026-07-05T11:09:01.473930+00:00"},{"alias_kind":"pith_short_8","alias_value":"4BXPMGPB","created_at":"2026-07-05T11:09:01.473930+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11740","citing_title":"UniReason-Med: A Shared Grounded Reasoning Interface for 2D-to-3D Transfer in Medical VQA","ref_index":100,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10819","citing_title":"Earth-OneVision: Extending Remote Sensing Multimodal Large Language Models to More Sensor Modalities and Tasks","ref_index":127,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28920","citing_title":"ExACT: Exemplar-Driven Calibrated Refinement for Training-Free Visual Grounding in Remote Sensing Images","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22034","citing_title":"AgroVG: A Large-Scale Multi-Source Benchmark for Agricultural Visual Grounding","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2511.23332","citing_title":"UniGeoSeg: Towards Unified Open-World Segmentation for Geospatial Scenes","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23214","citing_title":"DARC-CLIP: Dynamic Adaptive Refinement with Cross-Attention for Meme Understanding","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07765","citing_title":"RemoteAgent: Bridging Vague Human Intents and Earth Observation with RL-based Agentic MLLMs","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4BXPMGPBDQ5RYRFUFTDOGHDOHD","json":"https://pith.science/pith/4BXPMGPBDQ5RYRFUFTDOGHDOHD.json","graph_json":"https://pith.science/api/pith-number/4BXPMGPBDQ5RYRFUFTDOGHDOHD/graph.json","events_json":"https://pith.science/api/pith-number/4BXPMGPBDQ5RYRFUFTDOGHDOHD/events.json","paper":"https://pith.science/paper/4BXPMGPB"},"agent_actions":{"view_html":"https://pith.science/pith/4BXPMGPBDQ5RYRFUFTDOGHDOHD","download_json":"https://pith.science/pith/4BXPMGPBDQ5RYRFUFTDOGHDOHD.json","view_paper":"https://pith.science/paper/4BXPMGPB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.08613&json=true","fetch_graph":"https://pith.science/api/pith-number/4BXPMGPBDQ5RYRFUFTDOGHDOHD/graph.json","fetch_events":"https://pith.science/api/pith-number/4BXPMGPBDQ5RYRFUFTDOGHDOHD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4BXPMGPBDQ5RYRFUFTDOGHDOHD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4BXPMGPBDQ5RYRFUFTDOGHDOHD/action/storage_attestation","attest_author":"https://pith.science/pith/4BXPMGPBDQ5RYRFUFTDOGHDOHD/action/author_attestation","sign_citation":"https://pith.science/pith/4BXPMGPBDQ5RYRFUFTDOGHDOHD/action/citation_signature","submit_replication":"https://pith.science/pith/4BXPMGPBDQ5RYRFUFTDOGHDOHD/action/replication_record"}},"created_at":"2026-07-05T11:09:01.473930+00:00","updated_at":"2026-07-05T11:09:01.473930+00:00"}