{"url":"/method/timesformer","slug":"timesformer","name":"TimeSformer","full_name":"TimeSformer","full_name_withheld":false,"description_markdown":"**TimeSformer** is a [convolution](https://paperswithcode.com/method/convolution)-free approach to video classification built exclusively on self-attention over space and time. It adapts the standard [Transformer](https://paperswithcode.com/method/transformer) architecture to video by enabling spatiotemporal feature learning directly from a sequence of frame-level patches. Specifically, the method adapts the image model [[Vision Transformer](https://paperswithcode.com/method/vision-transformer)](https//www.paperswithcode.com/method/vision-transformer) (ViT) to video by extending the self-attention mechanism from the image space to the space-time 3D volume. As in ViT, each patch is linearly mapped into an embedding and augmented with positional information. This makes it possible to interpret the resulting sequence of vector","description_state":"present","introduced_year":null,"introduced_by":{"title":"Is Space-Time Attention All You Need for Video Understanding?","paper":"/paper/is-space-time-attention-all-you-need-for","first_author":"Gedas Bertasius","n_authors":3,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/is-space-time-attention-all-you-need-for"},"source":{"url":"https://arxiv.org/abs/2102.05095v4","title":"Is Space-Time Attention All You Need for Video Understanding?","url_on_a_paper_host":true},"code_snippet_url":"https://github.com/pwc-1/paper1/tree/main/models/tapas","code_snippet_url_on_a_code_host":true,"categories":[{"area":"Computer Vision","area_id":"computer-vision","collection":"Generative Video Models","url":"/methods/category/generative-video-models","pwc_aliases":[]}],"n_papers_tagged":18,"archive_num_papers":18,"papers_newest_first":[{"paper":null,"title":"DualX-VSR: Dual Axial Spatial$\\times$Temporal Transformer for Real-World Video Super-Resolution without Motion Compensation","date":"2025-06-05","arxiv_id":"2506.04830","n_code_links":0,"syntology":null},{"paper":null,"title":"Fine-Tuning Video Transformers for Word-Level Bangla Sign Language: A Comparative Analysis for Classification Tasks","date":"2025-06-04","arxiv_id":"2506.04367","n_code_links":0,"syntology":null},{"paper":null,"title":"SkillFormer: Unified Multi-View Video Understanding for Proficiency Estimation","date":"2025-05-13","arxiv_id":"2505.08665","n_code_links":0,"syntology":null},{"paper":null,"title":"FedDA-TSformer: Federated Domain Adaptation with Vision TimeSformer for Left Ventricle Segmentation on Gated Myocardial Perfusion SPECT Image","date":"2025-02-23","arxiv_id":"2502.16709","n_code_links":0,"syntology":null},{"paper":null,"title":"EITNet: An IoT-Enhanced Framework for Real-Time Basketball Action Recognition","date":"2024-10-13","arxiv_id":"2410.09954","n_code_links":0,"syntology":null},{"paper":null,"title":"3D-LSPTM: An Automatic Framework with 3D-Large-Scale Pretrained Model for Laryngeal Cancer Detection Using Laryngoscopic Videos","date":"2024-09-02","arxiv_id":"2409.01459","n_code_links":0,"syntology":null},{"paper":"/paper/motion-meets-attention-video-motion-prompts","title":"Motion meets Attention: Video Motion Prompts","date":"2024-07-03","arxiv_id":"2407.03179","n_code_links":1,"syntology":{"ran":9,"of":11,"unverified":2,"pointer_only":0}},{"paper":null,"title":"Pig aggression classification using CNN, Transformers and Recurrent Networks","date":"2024-03-13","arxiv_id":"2403.08528","n_code_links":0,"syntology":null},{"paper":null,"title":"P-Age: Pexels Dataset for Robust Spatio-Temporal Apparent Age Classification","date":"2023-11-04","arxiv_id":"2311.02432","n_code_links":0,"syntology":null},{"paper":null,"title":"Scalable and Accurate Self-supervised Multimodal Representation Learning without Aligned Video and Text Data","date":"2023-04-04","arxiv_id":"2304.02080","n_code_links":0,"syntology":null},{"paper":null,"title":"Video Question Answering Using CLIP-Guided Visual-Text Attention","date":"2023-03-06","arxiv_id":"2303.03131","n_code_links":0,"syntology":null},{"paper":"/paper/cholectriplet2022-show-me-a-tool-and-tell-me","title":"CholecTriplet2022: Show me a tool and tell me the triplet -- an endoscopic vision challenge for surgical action triplet detection","date":"2023-02-13","arxiv_id":"2302.06294","n_code_links":2,"syntology":null},{"paper":"/paper/mintime-multi-identity-size-invariant-video","title":"MINTIME: Multi-Identity Size-Invariant Video Deepfake Detection","date":"2022-11-20","arxiv_id":"2211.10996","n_code_links":1,"syntology":null},{"paper":"/paper/one-model-is-not-enough-ensembles-for","title":"One Model is Not Enough: Ensembles for Isolated Sign Language Recognition","date":"2022-07-04","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"title":"Context-aware Proposal Network for Temporal Action Detection","date":"2022-06-18","arxiv_id":"2206.09082","n_code_links":0,"syntology":null},{"paper":"/paper/vidi-a-video-dataset-of-incidents","title":"VIDI: A Video Dataset of Incidents","date":"2022-05-26","arxiv_id":"2205.13277","n_code_links":1,"syntology":null},{"paper":"/paper/ia-red-2-interpretability-aware-redundancy","title":"IA-RED$^2$: Interpretability-Aware Redundancy Reduction for Vision Transformers","date":"2021-06-23","arxiv_id":"2106.12620","n_code_links":0,"syntology":null},{"paper":"/paper/is-space-time-attention-all-you-need-for","title":"Is Space-Time Attention All You Need for Video Understanding?","date":"2021-02-09","arxiv_id":"2102.05095","n_code_links":16,"syntology":{"ran":35,"of":43,"unverified":8,"pointer_only":14}}],"papers_shown":18,"tasks":[{"task":"/task/action-recognition-in-videos","name":"Action Recognition","papers":3},{"task":"/task/classification-1","name":"Classification","papers":3},{"task":"/task/video-understanding","name":"Video Understanding","papers":3},{"task":"/task/action-classification","name":"Action Classification","papers":2},{"task":"/task/sign-language-recognition","name":"Sign Language Recognition","papers":2},{"task":"/task/video-classification","name":"Video Classification","papers":2},{"task":"/task/video-question-answering","name":"Video Question Answering","papers":2},{"task":"/task/action-detection","name":"Action Detection","papers":1},{"task":"/task/action-triplet-detection","name":"Action Triplet Detection","papers":1},{"task":"/task/action-triplet-recognition","name":"Action Triplet Recognition","papers":1},{"task":"/task/age-classification","name":"Age Classification","papers":1},{"task":"/task/age-estimation","name":"Age Estimation","papers":1},{"task":"/task/all","name":"All","papers":1},{"task":"/task/anomaly-detection","name":"Anomaly Detection","papers":1},{"task":"/task/automatic-speech-recognition-2","name":"Automatic Speech Recognition","papers":1},{"task":"/task/automatic-speech-recognition","name":"Automatic Speech Recognition (ASR)","papers":1},{"task":"/task/computational-efficiency","name":"Computational Efficiency","papers":1},{"task":"/task/data-augmentation","name":"Data Augmentation","papers":1},{"task":"/task/deepfake-detection","name":"DeepFake Detection","papers":1},{"task":"/task/domain-adaptation","name":"Domain Adaptation","papers":1}],"tasks_shown":20,"n_tasks":46,"usage_by_year":[{"year":"2021","papers":2},{"year":"2022","papers":4},{"year":"2023","papers":4},{"year":"2024","papers":4},{"year":"2025","papers":4}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/timesformer"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}