{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:32DYMOFFNZ4ZEE3IF2OSFT7TBX","short_pith_number":"pith:32DYMOFF","schema_version":"1.0","canonical_sha256":"de878638a56e799213682e9d22cff30ddc9ea6baa4053010e4612ea4555487ff","source":{"kind":"arxiv","id":"2011.14503","version":5},"attestation_state":"computed","paper":{"title":"End-to-End Video Instance Segmentation with Transformers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Baoshan Cheng, Chunhua Shen, Hao Shen, Huaxia Xia, Xinlong Wang, Yuqing Wang, Zhaoliang Xu","submitted_at":"2020-11-30T02:03:50Z","abstract_excerpt":"Video instance segmentation (VIS) is the task that requires simultaneously classifying, segmenting and tracking object instances of interest in video. Recent methods typically develop sophisticated pipelines to tackle this task. Here, we propose a new video instance segmentation framework built upon Transformers, termed VisTR, which views the VIS task as a direct end-to-end parallel sequence decoding/prediction problem. Given a video clip consisting of multiple image frames as input, VisTR outputs the sequence of masks for each instance in the video in order directly. At the core is a new, eff"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2011.14503","kind":"arxiv","version":5},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2020-11-30T02:03:50Z","cross_cats_sorted":[],"title_canon_sha256":"dd854e0d391017fe8af3d7f08724ec615b8613404913cf18872a87c13dd5e79d","abstract_canon_sha256":"581109cc458b16dfd692431a643b1f564c6a63e9a8454ab9b199c25aa41d50ed"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:20:53.800451Z","signature_b64":"o9ehSeMgrGNCf95dtvvyNc4iRhfBSdyxKdTuvWDUZeUuK3YcrcgOSUEAuVVekStTQXamQCx/72g5By1hfaguAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"de878638a56e799213682e9d22cff30ddc9ea6baa4053010e4612ea4555487ff","last_reissued_at":"2026-07-05T03:20:53.800071Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:20:53.800071Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"End-to-End Video Instance Segmentation with Transformers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Baoshan Cheng, Chunhua Shen, Hao Shen, Huaxia Xia, Xinlong Wang, Yuqing Wang, Zhaoliang Xu","submitted_at":"2020-11-30T02:03:50Z","abstract_excerpt":"Video instance segmentation (VIS) is the task that requires simultaneously classifying, segmenting and tracking object instances of interest in video. Recent methods typically develop sophisticated pipelines to tackle this task. Here, we propose a new video instance segmentation framework built upon Transformers, termed VisTR, which views the VIS task as a direct end-to-end parallel sequence decoding/prediction problem. Given a video clip consisting of multiple image frames as input, VisTR outputs the sequence of masks for each instance in the video in order directly. At the core is a new, eff"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2011.14503","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2011.14503/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2011.14503","created_at":"2026-07-05T03:20:53.800131+00:00"},{"alias_kind":"arxiv_version","alias_value":"2011.14503v5","created_at":"2026-07-05T03:20:53.800131+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2011.14503","created_at":"2026-07-05T03:20:53.800131+00:00"},{"alias_kind":"pith_short_12","alias_value":"32DYMOFFNZ4Z","created_at":"2026-07-05T03:20:53.800131+00:00"},{"alias_kind":"pith_short_16","alias_value":"32DYMOFFNZ4ZEE3I","created_at":"2026-07-05T03:20:53.800131+00:00"},{"alias_kind":"pith_short_8","alias_value":"32DYMOFF","created_at":"2026-07-05T03:20:53.800131+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/32DYMOFFNZ4ZEE3IF2OSFT7TBX","json":"https://pith.science/pith/32DYMOFFNZ4ZEE3IF2OSFT7TBX.json","graph_json":"https://pith.science/api/pith-number/32DYMOFFNZ4ZEE3IF2OSFT7TBX/graph.json","events_json":"https://pith.science/api/pith-number/32DYMOFFNZ4ZEE3IF2OSFT7TBX/events.json","paper":"https://pith.science/paper/32DYMOFF"},"agent_actions":{"view_html":"https://pith.science/pith/32DYMOFFNZ4ZEE3IF2OSFT7TBX","download_json":"https://pith.science/pith/32DYMOFFNZ4ZEE3IF2OSFT7TBX.json","view_paper":"https://pith.science/paper/32DYMOFF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2011.14503&json=true","fetch_graph":"https://pith.science/api/pith-number/32DYMOFFNZ4ZEE3IF2OSFT7TBX/graph.json","fetch_events":"https://pith.science/api/pith-number/32DYMOFFNZ4ZEE3IF2OSFT7TBX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/32DYMOFFNZ4ZEE3IF2OSFT7TBX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/32DYMOFFNZ4ZEE3IF2OSFT7TBX/action/storage_attestation","attest_author":"https://pith.science/pith/32DYMOFFNZ4ZEE3IF2OSFT7TBX/action/author_attestation","sign_citation":"https://pith.science/pith/32DYMOFFNZ4ZEE3IF2OSFT7TBX/action/citation_signature","submit_replication":"https://pith.science/pith/32DYMOFFNZ4ZEE3IF2OSFT7TBX/action/replication_record"}},"created_at":"2026-07-05T03:20:53.800131+00:00","updated_at":"2026-07-05T03:20:53.800131+00:00"}