{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:ZBLCYUU3IXIHW7MMB5FBKOOQWR","short_pith_number":"pith:ZBLCYUU3","schema_version":"1.0","canonical_sha256":"c8562c529b45d07b7d8c0f4a1539d0b474d686e79c7ef13db402b955146ec0ad","source":{"kind":"arxiv","id":"2312.13305","version":1},"attestation_state":"computed","paper":{"title":"DVIS++: Improved Decoupled Framework for Universal Video Segmentation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Pengfei Wan, Shunping Ji, Tao Zhang, Xingye Tian, Xin Tao, Xuebo Wang, Yikang Zhou, Yuan Zhang, Yu Wu, Zhongyuan Wang","submitted_at":"2023-12-20T03:01:33Z","abstract_excerpt":"We present the \\textbf{D}ecoupled \\textbf{VI}deo \\textbf{S}egmentation (DVIS) framework, a novel approach for the challenging task of universal video segmentation, including video instance segmentation (VIS), video semantic segmentation (VSS), and video panoptic segmentation (VPS). Unlike previous methods that model video segmentation in an end-to-end manner, our approach decouples video segmentation into three cascaded sub-tasks: segmentation, tracking, and refinement. This decoupling design allows for simpler and more effective modeling of the spatio-temporal representations of objects, espe"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.13305","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-12-20T03:01:33Z","cross_cats_sorted":[],"title_canon_sha256":"ad7ff23c60c21fc89e97f7840e7fdce4d8751775864739cd9e10c97928c91e83","abstract_canon_sha256":"89c1e45cc48163caeb5624d44eab88da537b5a941a035f9d432674f8bf49c7cb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:26:43.136784Z","signature_b64":"OSrZR5Tjq7/G73fEaCMK3H/MdmdXXrZRCa3kIkjM0vdYvvCPrDcCfTamjtxvpwuGZ+ww/NRwOG5gDdSDn8MBCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c8562c529b45d07b7d8c0f4a1539d0b474d686e79c7ef13db402b955146ec0ad","last_reissued_at":"2026-07-05T07:26:43.136340Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:26:43.136340Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DVIS++: Improved Decoupled Framework for Universal Video Segmentation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Pengfei Wan, Shunping Ji, Tao Zhang, Xingye Tian, Xin Tao, Xuebo Wang, Yikang Zhou, Yuan Zhang, Yu Wu, Zhongyuan Wang","submitted_at":"2023-12-20T03:01:33Z","abstract_excerpt":"We present the \\textbf{D}ecoupled \\textbf{VI}deo \\textbf{S}egmentation (DVIS) framework, a novel approach for the challenging task of universal video segmentation, including video instance segmentation (VIS), video semantic segmentation (VSS), and video panoptic segmentation (VPS). Unlike previous methods that model video segmentation in an end-to-end manner, our approach decouples video segmentation into three cascaded sub-tasks: segmentation, tracking, and refinement. This decoupling design allows for simpler and more effective modeling of the spatio-temporal representations of objects, espe"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.13305","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.13305/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.13305","created_at":"2026-07-05T07:26:43.136392+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.13305v1","created_at":"2026-07-05T07:26:43.136392+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.13305","created_at":"2026-07-05T07:26:43.136392+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZBLCYUU3IXIH","created_at":"2026-07-05T07:26:43.136392+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZBLCYUU3IXIHW7MM","created_at":"2026-07-05T07:26:43.136392+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZBLCYUU3","created_at":"2026-07-05T07:26:43.136392+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.18010","citing_title":"Functionalization via Structure Completion and Motion Rectification","ref_index":214,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZBLCYUU3IXIHW7MMB5FBKOOQWR","json":"https://pith.science/pith/ZBLCYUU3IXIHW7MMB5FBKOOQWR.json","graph_json":"https://pith.science/api/pith-number/ZBLCYUU3IXIHW7MMB5FBKOOQWR/graph.json","events_json":"https://pith.science/api/pith-number/ZBLCYUU3IXIHW7MMB5FBKOOQWR/events.json","paper":"https://pith.science/paper/ZBLCYUU3"},"agent_actions":{"view_html":"https://pith.science/pith/ZBLCYUU3IXIHW7MMB5FBKOOQWR","download_json":"https://pith.science/pith/ZBLCYUU3IXIHW7MMB5FBKOOQWR.json","view_paper":"https://pith.science/paper/ZBLCYUU3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.13305&json=true","fetch_graph":"https://pith.science/api/pith-number/ZBLCYUU3IXIHW7MMB5FBKOOQWR/graph.json","fetch_events":"https://pith.science/api/pith-number/ZBLCYUU3IXIHW7MMB5FBKOOQWR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZBLCYUU3IXIHW7MMB5FBKOOQWR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZBLCYUU3IXIHW7MMB5FBKOOQWR/action/storage_attestation","attest_author":"https://pith.science/pith/ZBLCYUU3IXIHW7MMB5FBKOOQWR/action/author_attestation","sign_citation":"https://pith.science/pith/ZBLCYUU3IXIHW7MMB5FBKOOQWR/action/citation_signature","submit_replication":"https://pith.science/pith/ZBLCYUU3IXIHW7MMB5FBKOOQWR/action/replication_record"}},"created_at":"2026-07-05T07:26:43.136392+00:00","updated_at":"2026-07-05T07:26:43.136392+00:00"}