{"url":"/task/video-action-detection","name":"Video Action Detection","slug":"video-action-detection","description_markdown":null,"categories":[],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":32,"papers_with_code":19,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":1,"subtasks":0,"parent_tasks":0},"benchmarks":[],"datasets":[{"url":"/dataset/bah","name":"BAH","full_name":"Behavioural Ambivalence/Hesitancy","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":19,"of":19,"tagged_in_all":32,"items":[{"url":"/paper/context-aware-rcnn-a-baseline-for-action","title":"Context-Aware RCNN: A Baseline for Action Detection in Videos","date":"2020-07-20","arxiv_id":"2007.09861","repositories_listed":3,"syntology":null},{"url":"/paper/stable-mean-teacher-for-semi-supervised-video","title":"Stable Mean Teacher for Semi-supervised Video Action Detection","date":"2024-12-10","arxiv_id":"2412.07072","repositories_listed":2,"syntology":{"n":6,"n_ran":2,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/efficient-video-action-detection-with-token","title":"Efficient Video Action Detection with Token Dropout and Context Refinement","date":"2023-04-17","arxiv_id":"2304.08451","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":3}},{"url":"/paper/spotting-temporally-precise-fine-grained","title":"Spotting Temporally Precise, Fine-Grained Events in Video","date":"2022-07-20","arxiv_id":"2207.10213","repositories_listed":2,"syntology":{"n":5,"n_ran":1,"n_unverified":4,"n_pointer_only":1}},{"url":"/paper/asynchronous-interaction-aggregation-for","title":"Asynchronous Interaction Aggregation for Action Detection","date":"2020-04-16","arxiv_id":"2004.07485","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/actor-conditioned-attention-maps-for-video","title":"Actor Conditioned Attention Maps for Video Action Detection","date":"2018-12-30","arxiv_id":"1812.11631","repositories_listed":2,"syntology":null},{"url":"/paper/scaling-open-vocabulary-action-detection","title":"Scaling Open-Vocabulary Action Detection","date":"2025-04-04","arxiv_id":"2504.03096","repositories_listed":1,"syntology":null},{"url":"/paper/jovale-detecting-human-actions-in-video-using","title":"JoVALE: Detecting Human Actions in Video Using Audiovisual and Language Contexts","date":"2024-12-18","arxiv_id":"2412.13708","repositories_listed":1,"syntology":null},{"url":"/paper/on-occlusions-in-video-action-detection-1","title":"On Occlusions in Video Action Detection: Benchmark Datasets And Training Recipes","date":"2024-10-25","arxiv_id":"2410.19553","repositories_listed":1,"syntology":{"n":6,"n_ran":0,"n_unverified":6,"n_pointer_only":6}},{"url":"/paper/benchmarking-deep-learning-models-on-nvidia","title":"Benchmarking Deep Learning Models on NVIDIA Jetson Nano for Real-Time Systems: An Empirical Investigation","date":"2024-06-25","arxiv_id":"2406.17749","repositories_listed":1,"syntology":null},{"url":"/paper/generative-model-based-feature-knowledge","title":"Generative Model-based Feature Knowledge Distillation for Action Recognition","date":"2023-12-14","arxiv_id":"2312.08644","repositories_listed":1,"syntology":null},{"url":"/paper/semi-supervised-active-learning-for-video","title":"Semi-supervised Active Learning for Video Action Detection","date":"2023-12-12","arxiv_id":"2312.07169","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/e-2tad-an-energy-efficient-tracking-based","title":"E^2TAD: An Energy-Efficient Tracking-based Action Detector","date":"2022-04-09","arxiv_id":"2204.04416","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-semi-supervised-learning-for-video","title":"End-to-End Semi-Supervised Learning for Video Action Detection","date":"2022-03-08","arxiv_id":"2203.04251","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/a-stronger-baseline-for-ego-centric-action","title":"A Stronger Baseline for Ego-Centric Action Detection","date":"2021-06-13","arxiv_id":"2106.06942","repositories_listed":1,"syntology":null},{"url":"/paper/tuber-tube-transformer-for-action-detection","title":"TubeR: Tubelet Transformer for Video Action Detection","date":"2021-04-02","arxiv_id":"2104.00969","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/stage-spatio-temporal-attention-on-graph","title":"Video action detection by learning graph-based spatio-temporal interactions","date":"2019-12-09","arxiv_id":"1912.04316","repositories_listed":1,"syntology":null},{"url":"/paper/step-spatio-temporal-progressive-learning-for","title":"STEP: Spatio-Temporal Progressive Learning for Video Action Detection","date":"2019-04-19","arxiv_id":"1904.09288","repositories_listed":1,"syntology":null},{"url":"/paper/tube-convolutional-neural-network-t-cnn-for","title":"Tube Convolutional Neural Network (T-CNN) for Action Detection in Videos","date":"2017-03-30","arxiv_id":"1703.10664","repositories_listed":1,"syntology":null}],"syntology_records":8,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}