{"url":"/task/spatio-temporal-action-localization","name":"Spatio-Temporal Action Localization","slug":"spatio-temporal-action-localization","description_markdown":null,"categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":37,"papers_with_code":14,"benchmarks":1,"benchmark_tables_in_archive":1,"benchmark_tables_shown":1,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":6,"subtasks":0,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/spatio-temporal-action-localization-on-ava","slug":"spatio-temporal-action-localization-on-ava","dataset":"AVA-Kinetics","dataset_url":"/dataset/ava","rows_in_archive":7,"metrics":["val mAP","test mAP"],"first_row_in_archive_order":{"model":"VideoMAE V2-g","paper_title":"VideoMAE V2: Scaling Video Masked Autoencoders with Dual Masking","paper_url":"/paper/videomae-v2-scaling-video-masked-autoencoders","paper_date":"2023-03-29","arxiv_id":"2303.16727","code_links":[{"title":"OpenGVLab/VideoMAEv2","url":"https://github.com/OpenGVLab/VideoMAEv2"}],"syntology":{"n":6,"n_ran":2,"n_unverified":4,"n_pointer_only":0}}}],"datasets":[{"url":"/dataset/kinetics","name":"Kinetics","full_name":"Kinetics Human Action Video Dataset","num_papers_in_archive":1341},{"url":"/dataset/ava","name":"AVA","full_name":"Atomic Visual Actions","num_papers_in_archive":113},{"url":"/dataset/multisports","name":"MultiSports","full_name":"","num_papers_in_archive":20},{"url":"/dataset/vidhoi","name":"VidHOI","full_name":"","num_papers_in_archive":7},{"url":"/dataset/jrdb-act","name":"JRDB-Act","full_name":"","num_papers_in_archive":5},{"url":"/dataset/liris-human-activities-dataset","name":"LIRIS human activities dataset","full_name":"","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[{"url":"/task/action-localization","name":"Action Localization"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":14,"of":14,"tagged_in_all":37,"items":[{"url":"/paper/1st-place-solution-for-ava-kinetics-crossover","title":"1st place solution for AVA-Kinetics Crossover in AcitivityNet Challenge 2020","date":"2020-06-16","arxiv_id":"2006.09116","repositories_listed":3,"syntology":null},{"url":"/paper/actor-context-actor-relation-network-for","title":"Actor-Context-Actor Relation Network for Spatio-Temporal Action Localization","date":"2020-06-14","arxiv_id":"2006.07976","repositories_listed":3,"syntology":{"n":27,"n_ran":1,"n_unverified":26,"n_pointer_only":0}},{"url":"/paper/internvideo-general-video-foundation-models","title":"InternVideo: General Video Foundation Models via Generative and Discriminative Learning","date":"2022-12-06","arxiv_id":"2212.03191","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/contextualized-spatio-temporal-contrastive","title":"Contextualized Spatio-Temporal Contrastive Learning with Self-Supervision","date":"2021-12-09","arxiv_id":"2112.05181","repositories_listed":2,"syntology":null},{"url":"/paper/action-tubelet-detector-for-spatio-temporal","title":"Action Tubelet Detector for Spatio-Temporal Action Localization","date":"2017-05-04","arxiv_id":"1705.01861","repositories_listed":2,"syntology":null},{"url":"/paper/scaling-open-vocabulary-action-detection","title":"Scaling Open-Vocabulary Action Detection","date":"2025-04-04","arxiv_id":"2504.03096","repositories_listed":1,"syntology":null},{"url":"/paper/videomae-v2-scaling-video-masked-autoencoders","title":"VideoMAE V2: Scaling Video Masked Autoencoders with Dual Masking","date":"2023-03-29","arxiv_id":"2303.16727","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/unmasked-teacher-towards-training-efficient","title":"Unmasked Teacher: Towards Training-Efficient Video Foundation Models","date":"2023-03-28","arxiv_id":"2303.16058","repositories_listed":1,"syntology":{"n":8,"n_ran":3,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/e-2tad-an-energy-efficient-tracking-based","title":"E^2TAD: An Energy-Efficient Tracking-based Action Detector","date":"2022-04-09","arxiv_id":"2204.04416","repositories_listed":1,"syntology":null},{"url":"/paper/korsal-key-point-detection-based-online-real","title":"KORSAL: Key-point Detection based Online Real-Time Spatio-Temporal Action Localization","date":"2021-11-05","arxiv_id":"2111.03319","repositories_listed":1,"syntology":null},{"url":"/paper/st-hoi-a-spatial-temporal-baseline-for-human","title":"ST-HOI: A Spatial-Temporal Baseline for Human-Object Interaction Detection in Videos","date":"2021-05-25","arxiv_id":"2105.11731","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_unverified":3,"n_pointer_only":4}},{"url":"/paper/stage-spatio-temporal-attention-on-graph","title":"Video action detection by learning graph-based spatio-temporal interactions","date":"2019-12-09","arxiv_id":"1912.04316","repositories_listed":1,"syntology":null},{"url":"/paper/actor-centric-relation-network","title":"Actor-Centric Relation Network","date":"2018-07-28","arxiv_id":"1807.10982","repositories_listed":1,"syntology":null},{"url":"/paper/chained-multi-stream-networks-exploiting-pose","title":"Chained Multi-stream Networks Exploiting Pose, Motion, and Appearance for Action Classification and Detection","date":"2017-04-03","arxiv_id":"1704.00616","repositories_listed":1,"syntology":null}],"syntology_records":5,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}