{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:AH445RDSG54KILHVLGFNFSSZVQ","short_pith_number":"pith:AH445RDS","schema_version":"1.0","canonical_sha256":"01f9cec4723778a42cf5598ad2ca59ac3007127e3b91c50481ead67325a369de","source":{"kind":"arxiv","id":"2112.01526","version":2},"attestation_state":"computed","paper":{"title":"MViTv2: Improved Multiscale Vision Transformers for Classification and Detection","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bo Xiong, Chao-Yuan Wu, Christoph Feichtenhofer, Haoqi Fan, Jitendra Malik, Karttikeya Mangalam, Yanghao Li","submitted_at":"2021-12-02T18:59:57Z","abstract_excerpt":"In this paper, we study Multiscale Vision Transformers (MViTv2) as a unified architecture for image and video classification, as well as object detection. We present an improved version of MViT that incorporates decomposed relative positional embeddings and residual pooling connections. We instantiate this architecture in five sizes and evaluate it for ImageNet classification, COCO detection and Kinetics video recognition where it outperforms prior work. We further compare MViTv2s' pooling attention to window attention mechanisms where it outperforms the latter in accuracy/compute. Without bel"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2112.01526","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2021-12-02T18:59:57Z","cross_cats_sorted":[],"title_canon_sha256":"077ff074276839e3ab69963146efb991375fd33896a62de687e6eb297118abe2","abstract_canon_sha256":"032fc6ebda6c57bc58f350c3d4b85e63847b22b5aef05697450c8b4f9517fc6d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:09:59.558476Z","signature_b64":"nsWxx4Te5P7Jar9RyvnUHWd8RjWmCPd4pl9woB2Ck+Xe4fc0xm8v75kgsjHAqa+a66zM7Un4aYAMMGByog1cDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"01f9cec4723778a42cf5598ad2ca59ac3007127e3b91c50481ead67325a369de","last_reissued_at":"2026-07-05T04:09:59.558031Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:09:59.558031Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MViTv2: Improved Multiscale Vision Transformers for Classification and Detection","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bo Xiong, Chao-Yuan Wu, Christoph Feichtenhofer, Haoqi Fan, Jitendra Malik, Karttikeya Mangalam, Yanghao Li","submitted_at":"2021-12-02T18:59:57Z","abstract_excerpt":"In this paper, we study Multiscale Vision Transformers (MViTv2) as a unified architecture for image and video classification, as well as object detection. We present an improved version of MViT that incorporates decomposed relative positional embeddings and residual pooling connections. We instantiate this architecture in five sizes and evaluate it for ImageNet classification, COCO detection and Kinetics video recognition where it outperforms prior work. We further compare MViTv2s' pooling attention to window attention mechanisms where it outperforms the latter in accuracy/compute. Without bel"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2112.01526","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2112.01526/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2112.01526","created_at":"2026-07-05T04:09:59.558086+00:00"},{"alias_kind":"arxiv_version","alias_value":"2112.01526v2","created_at":"2026-07-05T04:09:59.558086+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2112.01526","created_at":"2026-07-05T04:09:59.558086+00:00"},{"alias_kind":"pith_short_12","alias_value":"AH445RDSG54K","created_at":"2026-07-05T04:09:59.558086+00:00"},{"alias_kind":"pith_short_16","alias_value":"AH445RDSG54KILHV","created_at":"2026-07-05T04:09:59.558086+00:00"},{"alias_kind":"pith_short_8","alias_value":"AH445RDS","created_at":"2026-07-05T04:09:59.558086+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.17133","citing_title":"CAM-VFD: Cross-Attention Multimodal Video Forgery Detection","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2312.14238","citing_title":"InternVL: Scaling up Vision Foundation Models and Aligning for Generic Visual-Linguistic Tasks","ref_index":86,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06783","citing_title":"Insights from Visual Cognition: Understanding Human Action Dynamics with Overall Glance and Refined Gaze Transformer","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02094","citing_title":"SignMAE: Segmentation-Driven Self-Supervised Learning for Sign Language Recognition","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AH445RDSG54KILHVLGFNFSSZVQ","json":"https://pith.science/pith/AH445RDSG54KILHVLGFNFSSZVQ.json","graph_json":"https://pith.science/api/pith-number/AH445RDSG54KILHVLGFNFSSZVQ/graph.json","events_json":"https://pith.science/api/pith-number/AH445RDSG54KILHVLGFNFSSZVQ/events.json","paper":"https://pith.science/paper/AH445RDS"},"agent_actions":{"view_html":"https://pith.science/pith/AH445RDSG54KILHVLGFNFSSZVQ","download_json":"https://pith.science/pith/AH445RDSG54KILHVLGFNFSSZVQ.json","view_paper":"https://pith.science/paper/AH445RDS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2112.01526&json=true","fetch_graph":"https://pith.science/api/pith-number/AH445RDSG54KILHVLGFNFSSZVQ/graph.json","fetch_events":"https://pith.science/api/pith-number/AH445RDSG54KILHVLGFNFSSZVQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AH445RDSG54KILHVLGFNFSSZVQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AH445RDSG54KILHVLGFNFSSZVQ/action/storage_attestation","attest_author":"https://pith.science/pith/AH445RDSG54KILHVLGFNFSSZVQ/action/author_attestation","sign_citation":"https://pith.science/pith/AH445RDSG54KILHVLGFNFSSZVQ/action/citation_signature","submit_replication":"https://pith.science/pith/AH445RDSG54KILHVLGFNFSSZVQ/action/replication_record"}},"created_at":"2026-07-05T04:09:59.558086+00:00","updated_at":"2026-07-05T04:09:59.558086+00:00"}