{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:3CZPLYOZHE6A4DMF3I5NNPGHJH","short_pith_number":"pith:3CZPLYOZ","schema_version":"1.0","canonical_sha256":"d8b2f5e1d9393c0e0d85da3ad6bcc749c1af97c53a55de765550c3916ca24dd6","source":{"kind":"arxiv","id":"2201.12288","version":2},"attestation_state":"computed","paper":{"title":"VRT: A Video Restoration Transformer","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["eess.IV"],"primary_cat":"cs.CV","authors_text":"Jiezhang Cao, Jingyun Liang, Kai Zhang, Luc Van Gool, Radu Timofte, Rakesh Ranjan, Yawei Li, Yuchen Fan","submitted_at":"2022-01-28T17:54:43Z","abstract_excerpt":"Video restoration (e.g., video super-resolution) aims to restore high-quality frames from low-quality frames. Different from single image restoration, video restoration generally requires to utilize temporal information from multiple adjacent but usually misaligned video frames. Existing deep methods generally tackle with this by exploiting a sliding window strategy or a recurrent architecture, which either is restricted by frame-by-frame restoration or lacks long-range modelling ability. In this paper, we propose a Video Restoration Transformer (VRT) with parallel frame prediction and long-ra"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2201.12288","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CV","submitted_at":"2022-01-28T17:54:43Z","cross_cats_sorted":["eess.IV"],"title_canon_sha256":"aa9443df58bdc2207e209afddfdb4937af091e17f7f99ab7077c33da8dab80c7","abstract_canon_sha256":"b312620d5066c6ea130488772dcda1fe09c953c9c5c0b9f43107f62447e6c6d1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:31:58.833664Z","signature_b64":"8KBGWK3llESpAo6NHeYVTOg0DFYGO+zK5bkfz4vIW92ouIGLsd5KaUAsbqrg1Jn/pODcU9zaH4xmGS8+QVRiDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d8b2f5e1d9393c0e0d85da3ad6bcc749c1af97c53a55de765550c3916ca24dd6","last_reissued_at":"2026-07-05T04:31:58.833185Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:31:58.833185Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VRT: A Video Restoration Transformer","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["eess.IV"],"primary_cat":"cs.CV","authors_text":"Jiezhang Cao, Jingyun Liang, Kai Zhang, Luc Van Gool, Radu Timofte, Rakesh Ranjan, Yawei Li, Yuchen Fan","submitted_at":"2022-01-28T17:54:43Z","abstract_excerpt":"Video restoration (e.g., video super-resolution) aims to restore high-quality frames from low-quality frames. Different from single image restoration, video restoration generally requires to utilize temporal information from multiple adjacent but usually misaligned video frames. Existing deep methods generally tackle with this by exploiting a sliding window strategy or a recurrent architecture, which either is restricted by frame-by-frame restoration or lacks long-range modelling ability. In this paper, we propose a Video Restoration Transformer (VRT) with parallel frame prediction and long-ra"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2201.12288","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2201.12288/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2201.12288","created_at":"2026-07-05T04:31:58.833242+00:00"},{"alias_kind":"arxiv_version","alias_value":"2201.12288v2","created_at":"2026-07-05T04:31:58.833242+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2201.12288","created_at":"2026-07-05T04:31:58.833242+00:00"},{"alias_kind":"pith_short_12","alias_value":"3CZPLYOZHE6A","created_at":"2026-07-05T04:31:58.833242+00:00"},{"alias_kind":"pith_short_16","alias_value":"3CZPLYOZHE6A4DMF","created_at":"2026-07-05T04:31:58.833242+00:00"},{"alias_kind":"pith_short_8","alias_value":"3CZPLYOZ","created_at":"2026-07-05T04:31:58.833242+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24336","citing_title":"TIGER: Taming Identity, Geometry, and Generative Priors for High-Quality Face Video Restoration","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.24336","citing_title":"TIGER: Taming Identity, Geometry, and Generative Priors for High-Quality Face Video Restoration","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2512.23709","citing_title":"Stream-DiffVSR: Low-Latency Streamable Video Super-Resolution via Auto-Regressive Diffusion","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2312.17090","citing_title":"Q-Align: Teaching LMMs for Visual Scoring via Discrete Text-Defined Levels","ref_index":97,"is_internal_anchor":false},{"citing_arxiv_id":"2603.22768","citing_title":"From Pixels to Semantics: A Multi-Stage AI Framework for Structural Damage Detection in Satellite Imagery","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3CZPLYOZHE6A4DMF3I5NNPGHJH","json":"https://pith.science/pith/3CZPLYOZHE6A4DMF3I5NNPGHJH.json","graph_json":"https://pith.science/api/pith-number/3CZPLYOZHE6A4DMF3I5NNPGHJH/graph.json","events_json":"https://pith.science/api/pith-number/3CZPLYOZHE6A4DMF3I5NNPGHJH/events.json","paper":"https://pith.science/paper/3CZPLYOZ"},"agent_actions":{"view_html":"https://pith.science/pith/3CZPLYOZHE6A4DMF3I5NNPGHJH","download_json":"https://pith.science/pith/3CZPLYOZHE6A4DMF3I5NNPGHJH.json","view_paper":"https://pith.science/paper/3CZPLYOZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2201.12288&json=true","fetch_graph":"https://pith.science/api/pith-number/3CZPLYOZHE6A4DMF3I5NNPGHJH/graph.json","fetch_events":"https://pith.science/api/pith-number/3CZPLYOZHE6A4DMF3I5NNPGHJH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3CZPLYOZHE6A4DMF3I5NNPGHJH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3CZPLYOZHE6A4DMF3I5NNPGHJH/action/storage_attestation","attest_author":"https://pith.science/pith/3CZPLYOZHE6A4DMF3I5NNPGHJH/action/author_attestation","sign_citation":"https://pith.science/pith/3CZPLYOZHE6A4DMF3I5NNPGHJH/action/citation_signature","submit_replication":"https://pith.science/pith/3CZPLYOZHE6A4DMF3I5NNPGHJH/action/replication_record"}},"created_at":"2026-07-05T04:31:58.833242+00:00","updated_at":"2026-07-05T04:31:58.833242+00:00"}