{"url":"/method/spatial-temporal-attention","slug":"spatial-temporal-attention","name":"Spatial & Temporal Attention","full_name":"Spatial & Temporal Attention","full_name_withheld":false,"description_markdown":"Spatial & temporal attention combines the advantages of spatial attention and temporal attention as it adaptively selects both important regions and key frames. Some works compute temporal attention and spatial attention separately, while others produce joint spatio & temporal attention maps. Further works focusing on capturing pairwise relations.","description_state":"present","introduced_year":null,"introduced_by":{"title":"An End-to-End Spatio-Temporal Attention Model for Human Action Recognition from Skeleton Data","paper":"/paper/an-end-to-end-spatio-temporal-attention-model","first_author":"Sijie Song","n_authors":5,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/an-end-to-end-spatio-temporal-attention-model"},"source":{"url":"http://arxiv.org/abs/1611.06067v1","title":"An End-to-End Spatio-Temporal Attention Model for Human Action Recognition from Skeleton Data","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"General","area_id":"general","collection":"Attention Mechanisms","url":"/methods/category/attention-mechanisms","pwc_aliases":["attention-mechanisms-1"]}],"n_papers_tagged":3,"archive_num_papers":3,"papers_newest_first":[{"paper":"/paper/explicitly-incorporating-spatial-information","title":"Explicitly incorporating spatial information to recurrent networks for agriculture","date":"2022-06-27","arxiv_id":"2206.13406","n_code_links":1,"syntology":null},{"paper":null,"title":"Retrieving and Highlighting Action with Spatiotemporal Reference","date":"2020-05-19","arxiv_id":"2005.09183","n_code_links":0,"syntology":null},{"paper":"/paper/an-end-to-end-spatio-temporal-attention-model","title":"An End-to-End Spatio-Temporal Attention Model for Human Action Recognition from Skeleton Data","date":"2016-11-18","arxiv_id":"1611.06067","n_code_links":0,"syntology":null}],"papers_shown":3,"tasks":[{"task":"/task/action-recognition-in-videos","name":"Action Recognition","papers":2},{"task":"/task/cross-modal-retrieval","name":"Cross-Modal Retrieval","papers":1},{"task":"/task/explainable-models","name":"Explainable Models","papers":1},{"task":"/task/retrieval","name":"Retrieval","papers":1},{"task":"/task/segmentation","name":"Segmentation","papers":1},{"task":"/task/semantic-segmentation","name":"Semantic Segmentation","papers":1},{"task":"/task/skeleton-based-action-recognition","name":"Skeleton Based Action Recognition","papers":1},{"task":"/task/action-recognition","name":"Temporal Action Localization","papers":1},{"task":"/task/text-to-video-retrieval","name":"Text to Video Retrieval","papers":1},{"task":"/task/video-text-retrieval","name":"Video-Text Retrieval","papers":1},{"task":"/task/visual-reasoning","name":"Visual Reasoning","papers":1},{"task":"/task/image-classification","name":"image-classification","papers":1}],"tasks_shown":12,"n_tasks":12,"usage_by_year":[{"year":"2016","papers":1},{"year":"2020","papers":1},{"year":"2022","papers":1}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/spatial-temporal-attention"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}