{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:N5DUEMWQGAO7NHBM4B7SXBYYEZ","short_pith_number":"pith:N5DUEMWQ","schema_version":"1.0","canonical_sha256":"6f474232d0301df69c2ce07f2b87182640cb5e3a56ddb20816a1aa58b6f608b6","source":{"kind":"arxiv","id":"2112.10764","version":1},"attestation_state":"computed","paper":{"title":"Mask2Former for Video Instance Segmentation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Alexander G. Schwing, Alexander Kirillov, Anwesa Choudhuri, Bowen Cheng, Ishan Misra, Rohit Girdhar","submitted_at":"2021-12-20T18:59:59Z","abstract_excerpt":"We find Mask2Former also achieves state-of-the-art performance on video instance segmentation without modifying the architecture, the loss or even the training pipeline. In this report, we show universal image segmentation architectures trivially generalize to video segmentation by directly predicting 3D segmentation volumes. Specifically, Mask2Former sets a new state-of-the-art of 60.4 AP on YouTubeVIS-2019 and 52.6 AP on YouTubeVIS-2021. We believe Mask2Former is also capable of handling video semantic and panoptic segmentation, given its versatility in image segmentation. We hope this will "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2112.10764","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2021-12-20T18:59:59Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"917b5e45644cbb39c0921e350a4df562f130326f53e457dde5b9d0d7603eac95","abstract_canon_sha256":"1822c677edb092c52c8a9892574597432d9e01be4183af74b847b1c80bd84dc2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:42:18.692776Z","signature_b64":"AtpeATQXOUUw0y4gV/Aut271pDpjCnaOQ3sY2Dvkpxw4pDeb1SZDZGrRp8b0u6kivHpmhVFURlj3e/mrE/63BA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6f474232d0301df69c2ce07f2b87182640cb5e3a56ddb20816a1aa58b6f608b6","last_reissued_at":"2026-07-05T03:42:18.692092Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:42:18.692092Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Mask2Former for Video Instance Segmentation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Alexander G. Schwing, Alexander Kirillov, Anwesa Choudhuri, Bowen Cheng, Ishan Misra, Rohit Girdhar","submitted_at":"2021-12-20T18:59:59Z","abstract_excerpt":"We find Mask2Former also achieves state-of-the-art performance on video instance segmentation without modifying the architecture, the loss or even the training pipeline. In this report, we show universal image segmentation architectures trivially generalize to video segmentation by directly predicting 3D segmentation volumes. Specifically, Mask2Former sets a new state-of-the-art of 60.4 AP on YouTubeVIS-2019 and 52.6 AP on YouTubeVIS-2021. We believe Mask2Former is also capable of handling video semantic and panoptic segmentation, given its versatility in image segmentation. We hope this will "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2112.10764","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2112.10764/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2112.10764","created_at":"2026-07-05T03:42:18.692168+00:00"},{"alias_kind":"arxiv_version","alias_value":"2112.10764v1","created_at":"2026-07-05T03:42:18.692168+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2112.10764","created_at":"2026-07-05T03:42:18.692168+00:00"},{"alias_kind":"pith_short_12","alias_value":"N5DUEMWQGAO7","created_at":"2026-07-05T03:42:18.692168+00:00"},{"alias_kind":"pith_short_16","alias_value":"N5DUEMWQGAO7NHBM","created_at":"2026-07-05T03:42:18.692168+00:00"},{"alias_kind":"pith_short_8","alias_value":"N5DUEMWQ","created_at":"2026-07-05T03:42:18.692168+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08688","citing_title":"SAM-MT: Real-Time Interactive Multi-Target Video Segmentation","ref_index":49,"is_internal_anchor":true},{"citing_arxiv_id":"2606.20140","citing_title":"SA-VIS: Sparse frame Annotations for training Video Instance Segmentation","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07394","citing_title":"Mind the Gap: Disentangling Performance Bottlenecks in Video Instance Segmentation","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20140","citing_title":"SA-VIS: Sparse frame Annotations for training Video Instance Segmentation","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2503.01835","citing_title":"Primus: Enforcing Attention Usage for 3D Medical Image Segmentation","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19411","citing_title":"GOLD-BEV: GrOund and aeriaL Data for Dense Semantic BEV Mapping of Dynamic Scenes","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13294","citing_title":"PAT-VCM: Plug-and-Play Auxiliary Tokens for Video Coding for Machines","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/N5DUEMWQGAO7NHBM4B7SXBYYEZ","json":"https://pith.science/pith/N5DUEMWQGAO7NHBM4B7SXBYYEZ.json","graph_json":"https://pith.science/api/pith-number/N5DUEMWQGAO7NHBM4B7SXBYYEZ/graph.json","events_json":"https://pith.science/api/pith-number/N5DUEMWQGAO7NHBM4B7SXBYYEZ/events.json","paper":"https://pith.science/paper/N5DUEMWQ"},"agent_actions":{"view_html":"https://pith.science/pith/N5DUEMWQGAO7NHBM4B7SXBYYEZ","download_json":"https://pith.science/pith/N5DUEMWQGAO7NHBM4B7SXBYYEZ.json","view_paper":"https://pith.science/paper/N5DUEMWQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2112.10764&json=true","fetch_graph":"https://pith.science/api/pith-number/N5DUEMWQGAO7NHBM4B7SXBYYEZ/graph.json","fetch_events":"https://pith.science/api/pith-number/N5DUEMWQGAO7NHBM4B7SXBYYEZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/N5DUEMWQGAO7NHBM4B7SXBYYEZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/N5DUEMWQGAO7NHBM4B7SXBYYEZ/action/storage_attestation","attest_author":"https://pith.science/pith/N5DUEMWQGAO7NHBM4B7SXBYYEZ/action/author_attestation","sign_citation":"https://pith.science/pith/N5DUEMWQGAO7NHBM4B7SXBYYEZ/action/citation_signature","submit_replication":"https://pith.science/pith/N5DUEMWQGAO7NHBM4B7SXBYYEZ/action/replication_record"}},"created_at":"2026-07-05T03:42:18.692168+00:00","updated_at":"2026-07-05T03:42:18.692168+00:00"}