{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:ZMAETS7E7TYIDUHMZXIEBJC2PI","short_pith_number":"pith:ZMAETS7E","schema_version":"1.0","canonical_sha256":"cb0049cbe4fcf081d0eccdd040a45a7a30bca682fcc58036eef241ab5618191a","source":{"kind":"arxiv","id":"2102.05095","version":4},"attestation_state":"computed","paper":{"title":"Is Space-Time Attention All You Need for Video Understanding?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Gedas Bertasius, Heng Wang, Lorenzo Torresani","submitted_at":"2021-02-09T19:49:33Z","abstract_excerpt":"We present a convolution-free approach to video classification built exclusively on self-attention over space and time. Our method, named \"TimeSformer,\" adapts the standard Transformer architecture to video by enabling spatiotemporal feature learning directly from a sequence of frame-level patches. Our experimental study compares different self-attention schemes and suggests that \"divided attention,\" where temporal attention and spatial attention are separately applied within each block, leads to the best video classification accuracy among the design choices considered. Despite the radically "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2102.05095","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2021-02-09T19:49:33Z","cross_cats_sorted":[],"title_canon_sha256":"bdb745219e0c5f01c11a53fa63f0779bc0fb937cc9d70a66a3b72516d8fc038e","abstract_canon_sha256":"40718f9070856e410a1f69cb3f81d201d7797481f33ac115acad7f674e47de81"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:47:42.906019Z","signature_b64":"fw1J/g8sAzzJb/95nuicb+dUB2UaYCzud08+SXHe5WSYzLl0awAj9E6aB9hU48HoQJRUFgqktA5XW4FhpdJfBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cb0049cbe4fcf081d0eccdd040a45a7a30bca682fcc58036eef241ab5618191a","last_reissued_at":"2026-07-05T02:47:42.905599Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:47:42.905599Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Is Space-Time Attention All You Need for Video Understanding?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Gedas Bertasius, Heng Wang, Lorenzo Torresani","submitted_at":"2021-02-09T19:49:33Z","abstract_excerpt":"We present a convolution-free approach to video classification built exclusively on self-attention over space and time. Our method, named \"TimeSformer,\" adapts the standard Transformer architecture to video by enabling spatiotemporal feature learning directly from a sequence of frame-level patches. Our experimental study compares different self-attention schemes and suggests that \"divided attention,\" where temporal attention and spatial attention are separately applied within each block, leads to the best video classification accuracy among the design choices considered. Despite the radically "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2102.05095","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2102.05095/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2102.05095","created_at":"2026-07-05T02:47:42.905667+00:00"},{"alias_kind":"arxiv_version","alias_value":"2102.05095v4","created_at":"2026-07-05T02:47:42.905667+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2102.05095","created_at":"2026-07-05T02:47:42.905667+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZMAETS7E7TYI","created_at":"2026-07-05T02:47:42.905667+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZMAETS7E7TYIDUHM","created_at":"2026-07-05T02:47:42.905667+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZMAETS7E","created_at":"2026-07-05T02:47:42.905667+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.04833","citing_title":"Signed Dual Attention: Capturing Signed Dependencies in Time Series Forecasting","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27686","citing_title":"Tensor Memory: Fixed-Size Recurrent State for Long-Horizon Transformers","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26260","citing_title":"A multi-task spatiotemporal deep neural network for predicting penetration depth and morphology in laser welding","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17133","citing_title":"CAM-VFD: Cross-Attention Multimodal Video Forgery Detection","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2506.19591","citing_title":"Vision Transformer-Based Time-Series Image Reconstruction for Cloud-Filling Applications","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2510.16371","citing_title":"Cataract-LMM Large-Scale Multi-Source Multi-Task Benchmark for Deep Learning in Surgical Video Analysis","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2512.10248","citing_title":"RobustSora: De-Watermarked Benchmark for Robust AI-Generated Video Detection","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2204.03458","citing_title":"Video Diffusion Models","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03848","citing_title":"Parameter-Efficient Multi-View Proficiency Estimation: From Discriminative Classification to Generative Feedback","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21780","citing_title":"Only Brains Align with Brains: Cross-Region Alignment Patterns Expose Limits of Normative Models","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06809","citing_title":"LookWhen? Fast Video Recognition by Learning When, Where, and What to Compute","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13279","citing_title":"Explainable Fall Detection for Elderly Monitoring via Temporally Stable SHAP in Skeleton-Based Human Activity Recognition","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20760","citing_title":"Exploring High-Order Self-Similarity for Video Understanding","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20311","citing_title":"Seeing Further and Wider: Joint Spatio-Temporal Enlargement for Micro-Video Popularity Prediction","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZMAETS7E7TYIDUHMZXIEBJC2PI","json":"https://pith.science/pith/ZMAETS7E7TYIDUHMZXIEBJC2PI.json","graph_json":"https://pith.science/api/pith-number/ZMAETS7E7TYIDUHMZXIEBJC2PI/graph.json","events_json":"https://pith.science/api/pith-number/ZMAETS7E7TYIDUHMZXIEBJC2PI/events.json","paper":"https://pith.science/paper/ZMAETS7E"},"agent_actions":{"view_html":"https://pith.science/pith/ZMAETS7E7TYIDUHMZXIEBJC2PI","download_json":"https://pith.science/pith/ZMAETS7E7TYIDUHMZXIEBJC2PI.json","view_paper":"https://pith.science/paper/ZMAETS7E","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2102.05095&json=true","fetch_graph":"https://pith.science/api/pith-number/ZMAETS7E7TYIDUHMZXIEBJC2PI/graph.json","fetch_events":"https://pith.science/api/pith-number/ZMAETS7E7TYIDUHMZXIEBJC2PI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZMAETS7E7TYIDUHMZXIEBJC2PI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZMAETS7E7TYIDUHMZXIEBJC2PI/action/storage_attestation","attest_author":"https://pith.science/pith/ZMAETS7E7TYIDUHMZXIEBJC2PI/action/author_attestation","sign_citation":"https://pith.science/pith/ZMAETS7E7TYIDUHMZXIEBJC2PI/action/citation_signature","submit_replication":"https://pith.science/pith/ZMAETS7E7TYIDUHMZXIEBJC2PI/action/replication_record"}},"created_at":"2026-07-05T02:47:42.905667+00:00","updated_at":"2026-07-05T02:47:42.905667+00:00"}