{"url":"/task/action-detection","name":"Action Detection","slug":"action-detection","description_markdown":"Action Detection aims to find both where and when an action occurs within a video clip and classify what the action is taking place. Typically results are given in the form of action tublets, which are action bounding boxes linked across time in the video. This is related to temporal localization, which seeks to identify the start and end frame of an action, and action recognition,  which seeks only to classify which action is taking place and typically assumes a trimmed video.","categories":[{"name":"Computer Vision","url":"/area/computer-vision"},{"name":"Time Series","url":"/area/time-series"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":817,"papers_with_code":277,"benchmarks":11,"benchmark_tables_in_archive":11,"benchmark_tables_shown":11,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":34,"subtasks":8,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/action-detection-on-ucf101-24","slug":"action-detection-on-ucf101-24","dataset":"UCF101-24","dataset_url":"/dataset/ucf101-24","rows_in_archive":19,"metrics":["Frame-mAP 0.5","Video-mAP 0.1","Video-mAP 0.2","Video-mAP 0.5"],"first_row_in_archive_order":{"model":"STAR/L","paper_title":"End-to-End Spatio-Temporal Action Localisation with Video Transformers","paper_url":"/paper/end-to-end-spatio-temporal-action","paper_date":"2023-04-24","arxiv_id":"2304.12160","code_links":[],"syntology":null}},{"leaderboard":"/sota/action-detection-on-j-hmdb","slug":"action-detection-on-j-hmdb","dataset":"J-HMDB","dataset_url":"/dataset/jhmdb","rows_in_archive":18,"metrics":["Frame-mAP 0.5","Video-mAP 0.2","Video-mAP 0.5"],"first_row_in_archive_order":{"model":"SiA","paper_title":"Scaling Open-Vocabulary Action Detection","paper_url":"/paper/scaling-open-vocabulary-action-detection","paper_date":"2025-04-04","arxiv_id":"2504.03096","code_links":[{"title":"siatheindochinese/sia_act_placeholder","url":"https://github.com/siatheindochinese/sia_act_placeholder"}],"syntology":null}},{"leaderboard":"/sota/action-detection-on-charades","slug":"action-detection-on-charades","dataset":"Charades","dataset_url":"/dataset/charades","rows_in_archive":16,"metrics":["mAP"],"first_row_in_archive_order":{"model":"TTM","paper_title":"Token Turing Machines","paper_url":"/paper/token-turing-machines","paper_date":"2022-11-16","arxiv_id":"2211.09119","code_links":[{"title":"google-research/scenic","url":"https://github.com/google-research/scenic"}],"syntology":null}},{"leaderboard":"/sota/action-detection-on-multi-thumos","slug":"action-detection-on-multi-thumos","dataset":"Multi-THUMOS","dataset_url":"/dataset/multithumos","rows_in_archive":8,"metrics":["mAP"],"first_row_in_archive_order":{"model":"MLAD","paper_title":"Modeling Multi-Label Action Dependencies for Temporal Action Localization","paper_url":"/paper/modeling-multi-label-action-dependencies-for","paper_date":"2021-03-04","arxiv_id":"2103.03027","code_links":[{"title":"ptirupat/MLAD","url":"https://github.com/ptirupat/MLAD"}],"syntology":{"n":5,"n_ran":5,"n_unverified":0,"n_pointer_only":5}}},{"leaderboard":"/sota/action-detection-on-ucf-sports","slug":"action-detection-on-ucf-sports","dataset":"UCF Sports","dataset_url":"/dataset/ucf-sports","rows_in_archive":7,"metrics":["Frame-mAP 0.5","Video-mAP 0.2","Video-mAP 0.5"],"first_row_in_archive_order":{"model":"T-CNN","paper_title":"Tube Convolutional Neural Network (T-CNN) for Action Detection in Videos","paper_url":"/paper/tube-convolutional-neural-network-t-cnn-for","paper_date":"2017-03-30","arxiv_id":"1703.10664","code_links":[{"title":"cyberpunk317/Action_detection","url":"https://github.com/cyberpunk317/Action_detection"}],"syntology":null}},{"leaderboard":"/sota/action-detection-on-thumos-14","slug":"action-detection-on-thumos-14","dataset":"THUMOS' 14","dataset_url":"/dataset/thumos14-1","rows_in_archive":4,"metrics":["mAP"],"first_row_in_archive_order":{"model":"MAT (Ours) Trans","paper_title":"Memory-and-Anticipation Transformer for Online Action Understanding","paper_url":"/paper/memory-and-anticipation-transformer-for","paper_date":"2023-08-15","arxiv_id":"2308.07893","code_links":[{"title":"echo0125/memory-and-anticipation-transformer","url":"https://github.com/echo0125/memory-and-anticipation-transformer"}],"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}}},{"leaderboard":"/sota/action-detection-on-multisports","slug":"action-detection-on-multisports","dataset":"MultiSports","dataset_url":"/dataset/multisports","rows_in_archive":2,"metrics":["Frame-mAP 0.5","Video-mAP 0.2","Video-mAP 0.5"],"first_row_in_archive_order":{"model":"HIT","paper_title":"Holistic Interaction Transformer Network for Action Detection","paper_url":"/paper/holistic-interaction-transformer-network-for","paper_date":"2022-10-23","arxiv_id":"2210.12686","code_links":[{"title":"joslefaure/hit","url":"https://github.com/joslefaure/hit"}],"syntology":null}},{"leaderboard":"/sota/action-detection-on-tsu","slug":"action-detection-on-tsu","dataset":"TSU","dataset_url":"/dataset/tsu","rows_in_archive":2,"metrics":["Frame-mAP"],"first_row_in_archive_order":{"model":"PDAN","paper_title":"PDAN: Pyramid Dilated Attention Network for Action Detection","paper_url":"/paper/pdan-pyramid-dilated-attention-network-for","paper_date":"2021-01-05","arxiv_id":null,"code_links":[{"title":"dairui01/PDAN","url":"https://github.com/dairui01/PDAN"}],"syntology":null}},{"leaderboard":"/sota/action-detection-on-ttstroke-21","slug":"action-detection-on-ttstroke-21","dataset":"TTStroke-21 ME22","dataset_url":"/dataset/ttstroke-21","rows_in_archive":2,"metrics":["IoU","mAP"],"first_row_in_archive_order":{"model":"STCNN-V2 (Vote decision)","paper_title":"Baseline Method for the Sport Task of MediaEval 2022 with 3D CNNs using Attention Mechanisms","paper_url":"/paper/baseline-method-for-the-sport-task-of","paper_date":"2023-02-06","arxiv_id":"2302.02752","code_links":[{"title":"ccp-eva/sporttaskme22","url":"https://github.com/ccp-eva/sporttaskme22"}],"syntology":null}},{"leaderboard":"/sota/action-detection-on-ttstroke-21-me21","slug":"action-detection-on-ttstroke-21-me21","dataset":"TTStroke-21 ME21","dataset_url":"/dataset/ttstroke-21-me21","rows_in_archive":2,"metrics":["IoU","mAP"],"first_row_in_archive_order":{"model":"STCNN","paper_title":"Spatio-Temporal CNN baseline method for the Sports Video Task of MediaEval 2021 benchmark","paper_url":"/paper/spatio-temporal-cnn-baseline-method-for-the","paper_date":"2021-12-16","arxiv_id":"2112.12074","code_links":[{"title":"ccp-eva/sporttaskme21","url":"https://github.com/ccp-eva/sporttaskme21"}],"syntology":null}},{"leaderboard":"/sota/action-detection-on-multithumos-1","slug":"action-detection-on-multithumos-1","dataset":"MultiTHUMOS","dataset_url":"/dataset/multithumos","rows_in_archive":1,"metrics":["mAP"],"first_row_in_archive_order":{"model":"PAT","paper_title":"PAT: Position-Aware Transformer for Dense Multi-Label Action Detection","paper_url":"/paper/pat-position-aware-transformer-for-dense","paper_date":"2023-08-09","arxiv_id":"2308.05051","code_links":[],"syntology":null}}],"datasets":[{"url":"/dataset/activitynet","name":"ActivityNet","full_name":"","num_papers_in_archive":807},{"url":"/dataset/charades","name":"Charades","full_name":"","num_papers_in_archive":428},{"url":"/dataset/thumos14-1","name":"THUMOS14","full_name":"","num_papers_in_archive":318},{"url":"/dataset/jhmdb","name":"JHMDB","full_name":"Joint-annotated Human Motion Data Base","num_papers_in_archive":249},{"url":"/dataset/ava","name":"AVA","full_name":"Atomic Visual Actions","num_papers_in_archive":113},{"url":"/dataset/multithumos","name":"MultiTHUMOS","full_name":"","num_papers_in_archive":58},{"url":"/dataset/toyota-smarthome","name":"Toyota Smarthome Dataset","full_name":"","num_papers_in_archive":31},{"url":"/dataset/tvseries","name":"TVSeries","full_name":"","num_papers_in_archive":29},{"url":"/dataset/road","name":"ROAD","full_name":"ROAD: The ROad event Awareness Dataset for Autonomous Driving","num_papers_in_archive":27},{"url":"/dataset/okutama-action","name":"Okutama-Action","full_name":"","num_papers_in_archive":24},{"url":"/dataset/multisports","name":"MultiSports","full_name":"","num_papers_in_archive":20},{"url":"/dataset/meva","name":"MEVA","full_name":"Multiview Extended Video with Activities","num_papers_in_archive":17},{"url":"/dataset/ucf101-24","name":"UCF101-24","full_name":"","num_papers_in_archive":17},{"url":"/dataset/tsu","name":"TSU","full_name":"Toyota Smarthome Untrimmed","num_papers_in_archive":16},{"url":"/dataset/ava-speech","name":"AVA-Speech","full_name":"","num_papers_in_archive":10},{"url":"/dataset/egoprocel","name":"EgoProceL","full_name":"","num_papers_in_archive":9},{"url":"/dataset/ucf-sports","name":"UCF Sports","full_name":"","num_papers_in_archive":9},{"url":"/dataset/urfd-dataset","name":"URFD Dataset","full_name":"UR Fall Detection Dataset","num_papers_in_archive":7},{"url":"/dataset/vidhoi","name":"VidHOI","full_name":"","num_papers_in_archive":7},{"url":"/dataset/iitb-corridor","name":"IITB Corridor","full_name":"","num_papers_in_archive":5},{"url":"/dataset/mlb-youtube-dataset","name":"MLB-YouTube Dataset","full_name":null,"num_papers_in_archive":5},{"url":"/dataset/mpii-cooking-2-dataset","name":"MPII Cooking 2 Dataset","full_name":"","num_papers_in_archive":5},{"url":"/dataset/real-life-violence-situations-dataset","name":"Real Life Violence Situations Dataset","full_name":"","num_papers_in_archive":5},{"url":"/dataset/wear","name":"WEAR","full_name":"WEAR: An Outdoor Sports Dataset for Wearable and Egocentric Activity Recognition","num_papers_in_archive":5},{"url":"/dataset/esad","name":"ESAD","full_name":"SARAS Endoscopic Surgeon Action Detection","num_papers_in_archive":4},{"url":"/dataset/oreba","name":"OREBA","full_name":"Objectively Recognizing Eating Behavior and Associated Intake","num_papers_in_archive":3},{"url":"/dataset/ttstroke-21-me21","name":"TTStroke-21 ME21","full_name":"TTStroke-21 for MediaEval 2021","num_papers_in_archive":3},{"url":"/dataset/ttstroke-21","name":"TTStroke-21 ME22","full_name":"TTStroke-21 for MediaEval 2022","num_papers_in_archive":3},{"url":"/dataset/ucla-protest-image","name":"UCLA Protest Image","full_name":"","num_papers_in_archive":2},{"url":"/dataset/capture-24","name":"Capture-24","full_name":"","num_papers_in_archive":1},{"url":"/dataset/custom-spatio-temporal-action-video-dataset","name":"Custom Spatio-Temporal Action Video Dataset","full_name":"","num_papers_in_archive":1},{"url":"/dataset/liris-human-activities-dataset","name":"LIRIS human activities dataset","full_name":"","num_papers_in_archive":1},{"url":"/dataset/mouse-grooming-behavior","name":"Mouse Grooming Behavior","full_name":"","num_papers_in_archive":1},{"url":"/dataset/infiniterep","name":"InfiniteRep","full_name":"InfiniteRep","num_papers_in_archive":0}],"subtasks":[{"url":"/task/action-triplet-detection","name":"Action Triplet Detection"},{"url":"/task/audio-visual-active-speaker-detection","name":"Audio-Visual Active Speaker Detection"},{"url":"/task/few-shot-temporal-action-localization","name":"Few Shot Temporal Action Localization"},{"url":"/task/fine-grained-action-detection","name":"Fine-Grained Action Detection"},{"url":"/task/human-activity-recognition","name":"Human Activity Recognition"},{"url":"/task/multiple-action-detection","name":"Multiple Action Detection"},{"url":"/task/online-action-detection","name":"Online Action Detection"},{"url":"/task/skeleton-based-action-recognition","name":"Skeleton Based Action Recognition"}],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":277,"tagged_in_all":817,"items":[{"url":"/paper/continuous-control-with-deep-reinforcement","title":"Continuous control with deep reinforcement learning","date":"2015-09-09","arxiv_id":"1509.02971","repositories_listed":161,"syntology":{"n":306,"n_ran":158,"n_unverified":148,"n_pointer_only":163}},{"url":"/paper/bsn-boundary-sensitive-network-for-temporal","title":"BSN: Boundary Sensitive Network for Temporal Action Proposal Generation","date":"2018-06-08","arxiv_id":"1806.02964","repositories_listed":17,"syntology":{"n":4,"n_ran":3,"n_unverified":1,"n_pointer_only":4}},{"url":"/paper/bmn-boundary-matching-network-for-temporal","title":"BMN: Boundary-Matching Network for Temporal Action Proposal Generation","date":"2019-07-23","arxiv_id":"1907.09702","repositories_listed":15,"syntology":{"n":11,"n_ran":3,"n_unverified":8,"n_pointer_only":0}},{"url":"/paper/slowfast-networks-for-video-recognition","title":"SlowFast Networks for Video Recognition","date":"2018-12-10","arxiv_id":"1812.03982","repositories_listed":15,"syntology":{"n":10,"n_ran":0,"n_unverified":10,"n_pointer_only":0}},{"url":"/paper/ava-a-video-dataset-of-spatio-temporally","title":"AVA: A Video Dataset of Spatio-temporally Localized Atomic Visual Actions","date":"2017-05-23","arxiv_id":"1705.08421","repositories_listed":9,"syntology":null},{"url":"/paper/rescaling-egocentric-vision","title":"Rescaling Egocentric Vision","date":"2020-06-23","arxiv_id":"2006.13256","repositories_listed":7,"syntology":{"n":18,"n_ran":3,"n_unverified":15,"n_pointer_only":0}},{"url":"/paper/cholectriplet2021-a-benchmark-challenge-for","title":"CholecTriplet2021: A benchmark challenge for surgical action triplet recognition","date":"2022-04-10","arxiv_id":"2204.04746","repositories_listed":6,"syntology":null},{"url":"/paper/temporal-action-detection-with-structured","title":"Temporal Action Detection with Structured Segment Networks","date":"2017-04-20","arxiv_id":"1704.06228","repositories_listed":6,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/you-only-watch-once-a-unified-cnn","title":"You Only Watch Once: A Unified CNN Architecture for Real-Time Spatiotemporal Action Localization","date":"2019-11-15","arxiv_id":"1911.06644","repositories_listed":5,"syntology":{"n":12,"n_ran":3,"n_unverified":9,"n_pointer_only":2}},{"url":"/paper/joint-analysis-and-prediction-of-human","title":"From Recognition to Prediction: Analysis of Human Action and Trajectory Prediction in Video","date":"2020-11-20","arxiv_id":"2011.10670","repositories_listed":4,"syntology":null},{"url":"/paper/hake-human-activity-knowledge-engine","title":"HAKE: Human Activity Knowledge Engine","date":"2019-04-13","arxiv_id":"1904.06539","repositories_listed":4,"syntology":null},{"url":"/paper/locally-consistent-deformable-convolution","title":"Learning Motion in Feature Space: Locally-Consistent Deformable Convolution Networks for Fine-Grained Action Detection","date":"2018-11-21","arxiv_id":"1811.08815","repositories_listed":4,"syntology":null},{"url":"/paper/moshi-a-speech-text-foundation-model-for-real","title":"Moshi: a speech-text foundation model for real-time dialogue","date":"2024-09-17","arxiv_id":"2410.00037","repositories_listed":3,"syntology":{"n":7,"n_ran":2,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/no-time-to-waste-squeeze-time-into-channel","title":"No Time to Waste: Squeeze Time into Channel for Mobile Video Understanding","date":"2024-05-14","arxiv_id":"2405.08344","repositories_listed":3,"syntology":null},{"url":"/paper/temporal-action-localization-with-enhanced","title":"Temporal Action Localization with Enhanced Instant Discriminability","date":"2023-09-11","arxiv_id":"2309.05590","repositories_listed":3,"syntology":null},{"url":"/paper/multi-speaker-and-wide-band-simulated","title":"Multi-Speaker and Wide-Band Simulated Conversations as Training Data for End-to-End Neural Diarization","date":"2022-11-12","arxiv_id":"2211.06750","repositories_listed":3,"syntology":null},{"url":"/paper/road-the-road-event-awareness-dataset-for","title":"ROAD: The ROad event Awareness Dataset for Autonomous Driving","date":"2021-02-23","arxiv_id":"2102.11585","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/context-aware-rcnn-a-baseline-for-action","title":"Context-Aware RCNN: A Baseline for Action Detection in Videos","date":"2020-07-20","arxiv_id":"2007.09861","repositories_listed":3,"syntology":null},{"url":"/paper/actor-context-actor-relation-network-for","title":"Actor-Context-Actor Relation Network for Spatio-Temporal Action Localization","date":"2020-06-14","arxiv_id":"2006.07976","repositories_listed":3,"syntology":{"n":27,"n_ran":1,"n_unverified":26,"n_pointer_only":0}},{"url":"/paper/a-multigrid-method-for-efficiently-training","title":"A Multigrid Method for Efficiently Training Video Models","date":"2019-12-02","arxiv_id":"1912.00998","repositories_listed":3,"syntology":{"n":10,"n_ran":2,"n_unverified":8,"n_pointer_only":3}},{"url":"/paper/pyannoteaudio-neural-building-blocks-for","title":"pyannote.audio: neural building blocks for speaker diarization","date":"2019-11-04","arxiv_id":"1911.01255","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/rvad-an-unsupervised-segment-based-robust","title":"rVAD: An Unsupervised Segment-Based Robust Voice Activity Detection Method","date":"2019-06-09","arxiv_id":"1906.03588","repositories_listed":3,"syntology":null},{"url":"/paper/fine-grained-activity-recognition-in-baseball","title":"Fine-grained Activity Recognition in Baseball Videos","date":"2018-04-09","arxiv_id":"1804.03247","repositories_listed":3,"syntology":null},{"url":"/paper/r-c3d-region-convolutional-3d-network-for","title":"R-C3D: Region Convolutional 3D Network for Temporal Activity Detection","date":"2017-03-22","arxiv_id":"1703.07814","repositories_listed":3,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/an-end-to-end-architecture-for-keyword","title":"An End-to-End Architecture for Keyword Spotting and Voice Activity Detection","date":"2016-11-28","arxiv_id":"1611.09405","repositories_listed":3,"syntology":null},{"url":"/paper/temporal-activity-detection-in-untrimmed","title":"Temporal Activity Detection in Untrimmed Videos with Recurrent Neural Networks","date":"2016-08-29","arxiv_id":"1608.08128","repositories_listed":3,"syntology":null},{"url":"/paper/stable-mean-teacher-for-semi-supervised-video","title":"Stable Mean Teacher for Semi-supervised Video Action Detection","date":"2024-12-10","arxiv_id":"2412.07072","repositories_listed":2,"syntology":{"n":6,"n_ran":2,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/action-ood-an-end-to-end-skeleton-based-model","title":"Skeleton-OOD: An End-to-End Skeleton-Based Model for Robust Out-of-Distribution Human Action Detection","date":"2024-05-31","arxiv_id":"2405.20633","repositories_listed":2,"syntology":null},{"url":"/paper/end-to-end-temporal-action-detection-with-1b","title":"End-to-End Temporal Action Detection with 1B Parameters Across 1000 Frames","date":"2023-11-28","arxiv_id":"2311.17241","repositories_listed":2,"syntology":null},{"url":"/paper/ivrit-ai-a-comprehensive-dataset-of-hebrew","title":"ivrit.ai: A Comprehensive Dataset of Hebrew Speech for AI Research and Development","date":"2023-07-17","arxiv_id":"2307.08720","repositories_listed":2,"syntology":null}],"syntology_records":14,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}