{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:RTS7P2PISXO44LLGLPYK4L4ULW","short_pith_number":"pith:RTS7P2PI","schema_version":"1.0","canonical_sha256":"8ce5f7e9e895ddce2d665bf0ae2f945d8eb8c506cea1befc973125934a8b4c74","source":{"kind":"arxiv","id":"2411.17606","version":2},"attestation_state":"computed","paper":{"title":"HyperSeg: Towards Universal Visual Segmentation with Large Language Model","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Cong Wei, Haoxian Tan, Jie Hu, Yong Liu, Yujie Zhong, Yujiu Yang, Zheng Zhao","submitted_at":"2024-11-26T17:18:20Z","abstract_excerpt":"This paper aims to address universal segmentation for image and video perception with the strong reasoning ability empowered by Visual Large Language Models (VLLMs). Despite significant progress in current unified segmentation methods, limitations in adaptation to both image and video scenarios, as well as the complex reasoning segmentation, make it difficult for them to handle various challenging instructions and achieve an accurate understanding of fine-grained vision-language correlations. We propose HyperSeg, the first VLLM-based universal segmentation model for pixel-level image and video"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.17606","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-11-26T17:18:20Z","cross_cats_sorted":[],"title_canon_sha256":"bb9214a2d0e6d1282c5d36936bd7116e0d7b1a693b7e650e6413e4db254117a2","abstract_canon_sha256":"0b2758e2a7bb5db940c9762997aac11178310cddd16010f9a13c7e18b9056c13"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:42:33.778570Z","signature_b64":"HysvLwHoOBUwPdPW8dDMRH52TcDPWR1aTju0f4TezpXgzrIegGTS07zuJfeT1+05ZgPvTY+3vOthGEfW89CFCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8ce5f7e9e895ddce2d665bf0ae2f945d8eb8c506cea1befc973125934a8b4c74","last_reissued_at":"2026-07-05T09:42:33.778029Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:42:33.778029Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"HyperSeg: Towards Universal Visual Segmentation with Large Language Model","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Cong Wei, Haoxian Tan, Jie Hu, Yong Liu, Yujie Zhong, Yujiu Yang, Zheng Zhao","submitted_at":"2024-11-26T17:18:20Z","abstract_excerpt":"This paper aims to address universal segmentation for image and video perception with the strong reasoning ability empowered by Visual Large Language Models (VLLMs). Despite significant progress in current unified segmentation methods, limitations in adaptation to both image and video scenarios, as well as the complex reasoning segmentation, make it difficult for them to handle various challenging instructions and achieve an accurate understanding of fine-grained vision-language correlations. We propose HyperSeg, the first VLLM-based universal segmentation model for pixel-level image and video"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.17606","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.17606/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.17606","created_at":"2026-07-05T09:42:33.778098+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.17606v2","created_at":"2026-07-05T09:42:33.778098+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.17606","created_at":"2026-07-05T09:42:33.778098+00:00"},{"alias_kind":"pith_short_12","alias_value":"RTS7P2PISXO4","created_at":"2026-07-05T09:42:33.778098+00:00"},{"alias_kind":"pith_short_16","alias_value":"RTS7P2PISXO44LLG","created_at":"2026-07-05T09:42:33.778098+00:00"},{"alias_kind":"pith_short_8","alias_value":"RTS7P2PI","created_at":"2026-07-05T09:42:33.778098+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09303","citing_title":"Reason Twice: Segmentation via Candidate Discovery and Comparative Reasoning","ref_index":74,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00115","citing_title":"PixelEyes: Decoupling Perception and Reasoning for Pinpoint Visual Evidence Seeking","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31924","citing_title":"InstanceControl: Controllable Complex Image Generation without Instance Labeling","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29461","citing_title":"FlowSeg: Dynamic Semantic Guidance for LLM-Conditioned Segmentation","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19410","citing_title":"Vision Harnessing Agent for Open Ad-hoc Segmentation","ref_index":80,"is_internal_anchor":false},{"citing_arxiv_id":"2506.21546","citing_title":"Counterfactual Segmentation Reasoning: Diagnosing and Mitigating Pixel-Grounding Hallucination","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2511.16719","citing_title":"SAM 3: Segment Anything with Concepts","ref_index":137,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00891","citing_title":"X2SAM: Any Segmentation in Images and Videos","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11789","citing_title":"LMMs Meet Object-Centric Vision: Understanding, Segmentation, Editing and Generation","ref_index":182,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07916","citing_title":"Tarot-SAM3: Training-free SAM3 for Any Referring Expression Segmentation","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07334","citing_title":"RCoT-Seg: Reinforced Chain-of-Thought for Video Reasoning and Segmentation","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22836","citing_title":"AgentRVOS for MeViS-Text Track of 5th PVUW Challenge: 3rd Method","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18665","citing_title":"APRVOS: 1st Place Winner of 5th PVUW MeViS-Audio Track","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RTS7P2PISXO44LLGLPYK4L4ULW","json":"https://pith.science/pith/RTS7P2PISXO44LLGLPYK4L4ULW.json","graph_json":"https://pith.science/api/pith-number/RTS7P2PISXO44LLGLPYK4L4ULW/graph.json","events_json":"https://pith.science/api/pith-number/RTS7P2PISXO44LLGLPYK4L4ULW/events.json","paper":"https://pith.science/paper/RTS7P2PI"},"agent_actions":{"view_html":"https://pith.science/pith/RTS7P2PISXO44LLGLPYK4L4ULW","download_json":"https://pith.science/pith/RTS7P2PISXO44LLGLPYK4L4ULW.json","view_paper":"https://pith.science/paper/RTS7P2PI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.17606&json=true","fetch_graph":"https://pith.science/api/pith-number/RTS7P2PISXO44LLGLPYK4L4ULW/graph.json","fetch_events":"https://pith.science/api/pith-number/RTS7P2PISXO44LLGLPYK4L4ULW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RTS7P2PISXO44LLGLPYK4L4ULW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RTS7P2PISXO44LLGLPYK4L4ULW/action/storage_attestation","attest_author":"https://pith.science/pith/RTS7P2PISXO44LLGLPYK4L4ULW/action/author_attestation","sign_citation":"https://pith.science/pith/RTS7P2PISXO44LLGLPYK4L4ULW/action/citation_signature","submit_replication":"https://pith.science/pith/RTS7P2PISXO44LLGLPYK4L4ULW/action/replication_record"}},"created_at":"2026-07-05T09:42:33.778098+00:00","updated_at":"2026-07-05T09:42:33.778098+00:00"}