{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:WWSBROR3ZQFHSXTETZTUJTPE7X","short_pith_number":"pith:WWSBROR3","schema_version":"1.0","canonical_sha256":"b5a418ba3bcc0a795e649e6744cde4fdc1a5b9e2145393441596151228bb8be5","source":{"kind":"arxiv","id":"2412.14006","version":1},"attestation_state":"computed","paper":{"title":"InstructSeg: Unifying Instructed Visual Segmentation with Multi-modal Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Cong Wei, Haoxian Tan, Yingsen Zeng, Yong Liu, Yujie Zhong, Yujiu Yang, Zheng Zhao","submitted_at":"2024-12-18T16:20:40Z","abstract_excerpt":"Boosted by Multi-modal Large Language Models (MLLMs), text-guided universal segmentation models for the image and video domains have made rapid progress recently. However, these methods are often developed separately for specific domains, overlooking the similarities in task settings and solutions across these two areas. In this paper, we define the union of referring segmentation and reasoning segmentation at both the image and video levels as Instructed Visual Segmentation (IVS). Correspondingly, we propose InstructSeg, an end-to-end segmentation pipeline equipped with MLLMs for IVS. Specifi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.14006","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-12-18T16:20:40Z","cross_cats_sorted":[],"title_canon_sha256":"c4d31fd5a720654145cf9318cfef6876667368e4164ee8c8073ca924b10b985f","abstract_canon_sha256":"9d190ab43e311f8c1f57aca4850c3ed1bfaa4feeaf881571cccd462e271f2ecd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:51:12.135215Z","signature_b64":"a4cJkm6ObKfp8BzsIykaILQCQ0RVKmjANYjAUqgJu4vVAJgzgu1g0S5H+iZSQmQ7uRhil69ZOKw7RBaPS6fsAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b5a418ba3bcc0a795e649e6744cde4fdc1a5b9e2145393441596151228bb8be5","last_reissued_at":"2026-07-05T09:51:12.134602Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:51:12.134602Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"InstructSeg: Unifying Instructed Visual Segmentation with Multi-modal Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Cong Wei, Haoxian Tan, Yingsen Zeng, Yong Liu, Yujie Zhong, Yujiu Yang, Zheng Zhao","submitted_at":"2024-12-18T16:20:40Z","abstract_excerpt":"Boosted by Multi-modal Large Language Models (MLLMs), text-guided universal segmentation models for the image and video domains have made rapid progress recently. However, these methods are often developed separately for specific domains, overlooking the similarities in task settings and solutions across these two areas. In this paper, we define the union of referring segmentation and reasoning segmentation at both the image and video levels as Instructed Visual Segmentation (IVS). Correspondingly, we propose InstructSeg, an end-to-end segmentation pipeline equipped with MLLMs for IVS. Specifi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.14006","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.14006/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.14006","created_at":"2026-07-05T09:51:12.134678+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.14006v1","created_at":"2026-07-05T09:51:12.134678+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.14006","created_at":"2026-07-05T09:51:12.134678+00:00"},{"alias_kind":"pith_short_12","alias_value":"WWSBROR3ZQFH","created_at":"2026-07-05T09:51:12.134678+00:00"},{"alias_kind":"pith_short_16","alias_value":"WWSBROR3ZQFHSXTE","created_at":"2026-07-05T09:51:12.134678+00:00"},{"alias_kind":"pith_short_8","alias_value":"WWSBROR3","created_at":"2026-07-05T09:51:12.134678+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26196","citing_title":"From Structure to Synergy: A Survey of Vision-Language Perception Paradigm Evolution in Multimodal Large Language Models","ref_index":80,"is_internal_anchor":false},{"citing_arxiv_id":"2506.21546","citing_title":"Counterfactual Segmentation Reasoning: Diagnosing and Mitigating Pixel-Grounding Hallucination","ref_index":42,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WWSBROR3ZQFHSXTETZTUJTPE7X","json":"https://pith.science/pith/WWSBROR3ZQFHSXTETZTUJTPE7X.json","graph_json":"https://pith.science/api/pith-number/WWSBROR3ZQFHSXTETZTUJTPE7X/graph.json","events_json":"https://pith.science/api/pith-number/WWSBROR3ZQFHSXTETZTUJTPE7X/events.json","paper":"https://pith.science/paper/WWSBROR3"},"agent_actions":{"view_html":"https://pith.science/pith/WWSBROR3ZQFHSXTETZTUJTPE7X","download_json":"https://pith.science/pith/WWSBROR3ZQFHSXTETZTUJTPE7X.json","view_paper":"https://pith.science/paper/WWSBROR3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.14006&json=true","fetch_graph":"https://pith.science/api/pith-number/WWSBROR3ZQFHSXTETZTUJTPE7X/graph.json","fetch_events":"https://pith.science/api/pith-number/WWSBROR3ZQFHSXTETZTUJTPE7X/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WWSBROR3ZQFHSXTETZTUJTPE7X/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WWSBROR3ZQFHSXTETZTUJTPE7X/action/storage_attestation","attest_author":"https://pith.science/pith/WWSBROR3ZQFHSXTETZTUJTPE7X/action/author_attestation","sign_citation":"https://pith.science/pith/WWSBROR3ZQFHSXTETZTUJTPE7X/action/citation_signature","submit_replication":"https://pith.science/pith/WWSBROR3ZQFHSXTETZTUJTPE7X/action/replication_record"}},"created_at":"2026-07-05T09:51:12.134678+00:00","updated_at":"2026-07-05T09:51:12.134678+00:00"}