{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:T562MNJBFITMXPVQ7LB26VWYID","short_pith_number":"pith:T562MNJB","schema_version":"1.0","canonical_sha256":"9f7da635212a26cbbeb0fac3af56d840c02d7a2649e568d7f3dda1ade9042c49","source":{"kind":"arxiv","id":"2508.00265","version":2},"attestation_state":"computed","paper":{"title":"Multimodal Referring Segmentation: A Survey","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chang Liu, Henghui Ding, Shuting He, Song Tang, Yu-Gang Jiang, Zuxuan Wu","submitted_at":"2025-08-01T02:14:00Z","abstract_excerpt":"Multimodal referring segmentation aims to segment target objects in visual scenes, such as images, videos, and 3D scenes, based on referring expressions in text or audio format. This task plays a crucial role in practical applications requiring accurate object perception based on user instructions. Over the past decade, it has gained significant attention in the multimodal community, driven by advances in convolutional neural networks, transformers, and large language models, all of which have substantially improved multimodal perception capabilities. This paper provides a comprehensive survey"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.00265","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-08-01T02:14:00Z","cross_cats_sorted":[],"title_canon_sha256":"20b12cca78b25b6b8d61131d04ebb3242956ef2262fce4e17a19df1742ffd23a","abstract_canon_sha256":"8b2ee99096a0bf387c9c86988b4377f106fbdb295673ecb1481c084a20ee3bd2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:48:52.175000Z","signature_b64":"0JFjMYpQiBfDpXXACn/sfZZcO1Dr3/A9g+6RaU5HdPwx/p3vHeOkWDB+Q05aiQaZ7HDw3Jr5wg6y6Rn6w148CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9f7da635212a26cbbeb0fac3af56d840c02d7a2649e568d7f3dda1ade9042c49","last_reissued_at":"2026-07-05T11:48:52.174515Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:48:52.174515Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Multimodal Referring Segmentation: A Survey","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chang Liu, Henghui Ding, Shuting He, Song Tang, Yu-Gang Jiang, Zuxuan Wu","submitted_at":"2025-08-01T02:14:00Z","abstract_excerpt":"Multimodal referring segmentation aims to segment target objects in visual scenes, such as images, videos, and 3D scenes, based on referring expressions in text or audio format. This task plays a crucial role in practical applications requiring accurate object perception based on user instructions. Over the past decade, it has gained significant attention in the multimodal community, driven by advances in convolutional neural networks, transformers, and large language models, all of which have substantially improved multimodal perception capabilities. This paper provides a comprehensive survey"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.00265","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.00265/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.00265","created_at":"2026-07-05T11:48:52.174575+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.00265v2","created_at":"2026-07-05T11:48:52.174575+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.00265","created_at":"2026-07-05T11:48:52.174575+00:00"},{"alias_kind":"pith_short_12","alias_value":"T562MNJBFITM","created_at":"2026-07-05T11:48:52.174575+00:00"},{"alias_kind":"pith_short_16","alias_value":"T562MNJBFITMXPVQ","created_at":"2026-07-05T11:48:52.174575+00:00"},{"alias_kind":"pith_short_8","alias_value":"T562MNJB","created_at":"2026-07-05T11:48:52.174575+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07433","citing_title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","ref_index":77,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23500","citing_title":"B-GRTO: Bootstrapped Group Relative Tool Optimization for Referring Segmentation","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23500","citing_title":"B-GRTO: Bootstrapped Group Relative Tool Optimization for Referring Segmentation","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2508.09977","citing_title":"A Survey on 3D Gaussian Splatting Applications: Segmentation, Editing, and Generation","ref_index":288,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02860","citing_title":"A Paradigm Shift: Fully End-to-End Training for Temporal Sentence Grounding in Videos","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14147","citing_title":"ROSE: Retrieval-Oriented Segmentation Enhancement","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/T562MNJBFITMXPVQ7LB26VWYID","json":"https://pith.science/pith/T562MNJBFITMXPVQ7LB26VWYID.json","graph_json":"https://pith.science/api/pith-number/T562MNJBFITMXPVQ7LB26VWYID/graph.json","events_json":"https://pith.science/api/pith-number/T562MNJBFITMXPVQ7LB26VWYID/events.json","paper":"https://pith.science/paper/T562MNJB"},"agent_actions":{"view_html":"https://pith.science/pith/T562MNJBFITMXPVQ7LB26VWYID","download_json":"https://pith.science/pith/T562MNJBFITMXPVQ7LB26VWYID.json","view_paper":"https://pith.science/paper/T562MNJB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.00265&json=true","fetch_graph":"https://pith.science/api/pith-number/T562MNJBFITMXPVQ7LB26VWYID/graph.json","fetch_events":"https://pith.science/api/pith-number/T562MNJBFITMXPVQ7LB26VWYID/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/T562MNJBFITMXPVQ7LB26VWYID/action/timestamp_anchor","attest_storage":"https://pith.science/pith/T562MNJBFITMXPVQ7LB26VWYID/action/storage_attestation","attest_author":"https://pith.science/pith/T562MNJBFITMXPVQ7LB26VWYID/action/author_attestation","sign_citation":"https://pith.science/pith/T562MNJBFITMXPVQ7LB26VWYID/action/citation_signature","submit_replication":"https://pith.science/pith/T562MNJBFITMXPVQ7LB26VWYID/action/replication_record"}},"created_at":"2026-07-05T11:48:52.174575+00:00","updated_at":"2026-07-05T11:48:52.174575+00:00"}