{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:TTOZQKVL7WHDN4JG5GJWHTF72P","short_pith_number":"pith:TTOZQKVL","schema_version":"1.0","canonical_sha256":"9cdd982aabfd8e36f126e99363ccbfd3f4437494bd4ddafd9b4aca492e2b0cb3","source":{"kind":"arxiv","id":"2106.13230","version":1},"attestation_state":"computed","paper":{"title":"Video Swin Transformer","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Han Hu, Jia Ning, Stephen Lin, Yixuan Wei, Yue Cao, Ze Liu, Zheng Zhang","submitted_at":"2021-06-24T17:59:46Z","abstract_excerpt":"The vision community is witnessing a modeling shift from CNNs to Transformers, where pure Transformer architectures have attained top accuracy on the major video recognition benchmarks. These video models are all built on Transformer layers that globally connect patches across the spatial and temporal dimensions. In this paper, we instead advocate an inductive bias of locality in video Transformers, which leads to a better speed-accuracy trade-off compared to previous approaches which compute self-attention globally even with spatial-temporal factorization. The locality of the proposed video a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2106.13230","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2021-06-24T17:59:46Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"1bc2a4bb413ef59f87703edc44c79c76db37bc5b89cc7341d5ee0e1848b01571","abstract_canon_sha256":"a5ca88bf05c2c62d831772f979e59566e6d014d8191466fb6106a3955fdad9db"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:52:10.782429Z","signature_b64":"7UnzhqatVhW6SaPy1h6+5bNFJqgTkuEfi11OlA5Syujx8ggZ/Z8itrnHWFv3sZFkl3DkxpOMulpIFw4WMIoBBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9cdd982aabfd8e36f126e99363ccbfd3f4437494bd4ddafd9b4aca492e2b0cb3","last_reissued_at":"2026-07-05T02:52:10.781951Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:52:10.781951Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Video Swin Transformer","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Han Hu, Jia Ning, Stephen Lin, Yixuan Wei, Yue Cao, Ze Liu, Zheng Zhang","submitted_at":"2021-06-24T17:59:46Z","abstract_excerpt":"The vision community is witnessing a modeling shift from CNNs to Transformers, where pure Transformer architectures have attained top accuracy on the major video recognition benchmarks. These video models are all built on Transformer layers that globally connect patches across the spatial and temporal dimensions. In this paper, we instead advocate an inductive bias of locality in video Transformers, which leads to a better speed-accuracy trade-off compared to previous approaches which compute self-attention globally even with spatial-temporal factorization. The locality of the proposed video a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2106.13230","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2106.13230/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2106.13230","created_at":"2026-07-05T02:52:10.782008+00:00"},{"alias_kind":"arxiv_version","alias_value":"2106.13230v1","created_at":"2026-07-05T02:52:10.782008+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2106.13230","created_at":"2026-07-05T02:52:10.782008+00:00"},{"alias_kind":"pith_short_12","alias_value":"TTOZQKVL7WHD","created_at":"2026-07-05T02:52:10.782008+00:00"},{"alias_kind":"pith_short_16","alias_value":"TTOZQKVL7WHDN4JG","created_at":"2026-07-05T02:52:10.782008+00:00"},{"alias_kind":"pith_short_8","alias_value":"TTOZQKVL","created_at":"2026-07-05T02:52:10.782008+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":16,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.17437","citing_title":"Spatio-Temporal Fusion Model for Standard View Classification of Echocardiographic Videos","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20693","citing_title":"Spatio-Temporal Wildfire Spread Prediction in Canada using a Video Swin-Hybrid-U-Net and Satellite Imagery","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12215","citing_title":"MLT-Dedup: Efficient Large-Scale Online Video Deduplication via Multi-Level Representations and Spatial-Temporal Matching","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05736","citing_title":"VTI-CoT: Visual-Textual Interleaved Chain of Thought for Video Reasoning","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27686","citing_title":"Tensor Memory: Fixed-Size Recurrent State for Long-Horizon Transformers","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2406.05615","citing_title":"Video-Language Understanding: A Survey from Model Architecture, Model Training, and Data Perspectives","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17133","citing_title":"CAM-VFD: Cross-Attention Multimodal Video Forgery Detection","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2509.21261","citing_title":"Every Subtlety Counts: Fine-grained Person Independence Micro-Action Recognition via Distributionally Robust Optimization","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2512.10248","citing_title":"RobustSora: De-Watermarked Benchmark for Robust AI-Generated Video Detection","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2111.11432","citing_title":"Florence: A New Foundation Model for Computer Vision","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08712","citing_title":"From Articulated Kinematics to Routed Visual Control for Action-Conditioned Surgical Video Generation","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08962","citing_title":"MegaScale-Omni: A Hyper-Scale, Workload-Resilient System for MultiModal LLM Training in Production","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11389","citing_title":"ConvFormer3D-TAP: Phase/Uncertainty-Aware Front-End Fusion for Cine CMR View Classification Pipelines","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09326","citing_title":"Multimodal Anomaly Detection for Human-Robot Interaction","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16987","citing_title":"DVAR: Adversarial Multi-Agent Debate for Video Authenticity Detection","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02094","citing_title":"SignMAE: Segmentation-Driven Self-Supervised Learning for Sign Language Recognition","ref_index":20,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TTOZQKVL7WHDN4JG5GJWHTF72P","json":"https://pith.science/pith/TTOZQKVL7WHDN4JG5GJWHTF72P.json","graph_json":"https://pith.science/api/pith-number/TTOZQKVL7WHDN4JG5GJWHTF72P/graph.json","events_json":"https://pith.science/api/pith-number/TTOZQKVL7WHDN4JG5GJWHTF72P/events.json","paper":"https://pith.science/paper/TTOZQKVL"},"agent_actions":{"view_html":"https://pith.science/pith/TTOZQKVL7WHDN4JG5GJWHTF72P","download_json":"https://pith.science/pith/TTOZQKVL7WHDN4JG5GJWHTF72P.json","view_paper":"https://pith.science/paper/TTOZQKVL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2106.13230&json=true","fetch_graph":"https://pith.science/api/pith-number/TTOZQKVL7WHDN4JG5GJWHTF72P/graph.json","fetch_events":"https://pith.science/api/pith-number/TTOZQKVL7WHDN4JG5GJWHTF72P/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TTOZQKVL7WHDN4JG5GJWHTF72P/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TTOZQKVL7WHDN4JG5GJWHTF72P/action/storage_attestation","attest_author":"https://pith.science/pith/TTOZQKVL7WHDN4JG5GJWHTF72P/action/author_attestation","sign_citation":"https://pith.science/pith/TTOZQKVL7WHDN4JG5GJWHTF72P/action/citation_signature","submit_replication":"https://pith.science/pith/TTOZQKVL7WHDN4JG5GJWHTF72P/action/replication_record"}},"created_at":"2026-07-05T02:52:10.782008+00:00","updated_at":"2026-07-05T02:52:10.782008+00:00"}