{"url":"/task/activity-detection","name":"Activity Detection","slug":"activity-detection","description_markdown":"Detecting activities in extended videos.","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":380,"papers_with_code":75,"benchmarks":1,"benchmark_tables_in_archive":1,"benchmark_tables_shown":1,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":12,"subtasks":0,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/activity-detection-on-ava-speech","slug":"activity-detection-on-ava-speech","dataset":"AVA-Speech","dataset_url":"/dataset/ava-speech","rows_in_archive":4,"metrics":["ROC-AUC"],"first_row_in_archive_order":{"model":"CNN-BiLSTM_best","paper_title":"A Hybrid CNN-BiLSTM Voice Activity Detector","paper_url":"/paper/a-hybrid-cnn-bilstm-voice-activity-detector","paper_date":"2021-03-05","arxiv_id":"2103.03529","code_links":[{"title":"NickWilkinson37/voxseg","url":"https://github.com/NickWilkinson37/voxseg"}],"syntology":null}}],"datasets":[{"url":"/dataset/ava","name":"AVA","full_name":"Atomic Visual Actions","num_papers_in_archive":113},{"url":"/dataset/toyota-smarthome","name":"Toyota Smarthome Dataset","full_name":"","num_papers_in_archive":31},{"url":"/dataset/road","name":"ROAD","full_name":"ROAD: The ROad event Awareness Dataset for Autonomous Driving","num_papers_in_archive":27},{"url":"/dataset/meva","name":"MEVA","full_name":"Multiview Extended Video with Activities","num_papers_in_archive":17},{"url":"/dataset/tsu","name":"TSU","full_name":"Toyota Smarthome Untrimmed","num_papers_in_archive":16},{"url":"/dataset/ava-speech","name":"AVA-Speech","full_name":"","num_papers_in_archive":10},{"url":"/dataset/home-action-genome","name":"Home Action Genome","full_name":"","num_papers_in_archive":9},{"url":"/dataset/iitb-corridor","name":"IITB Corridor","full_name":"","num_papers_in_archive":5},{"url":"/dataset/mlb-youtube-dataset","name":"MLB-YouTube Dataset","full_name":null,"num_papers_in_archive":5},{"url":"/dataset/ucla-protest-image","name":"UCLA Protest Image","full_name":"","num_papers_in_archive":2},{"url":"/dataset/dahlia-daily-human-life-activity","name":"DAHLIA","full_name":"DAily Human Life Activity","num_papers_in_archive":0},{"url":"/dataset/infiniterep","name":"InfiniteRep","full_name":"InfiniteRep","num_papers_in_archive":0}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":75,"tagged_in_all":380,"items":[{"url":"/paper/moshi-a-speech-text-foundation-model-for-real","title":"Moshi: a speech-text foundation model for real-time dialogue","date":"2024-09-17","arxiv_id":"2410.00037","repositories_listed":3,"syntology":{"n":7,"n_ran":2,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/multi-speaker-and-wide-band-simulated","title":"Multi-Speaker and Wide-Band Simulated Conversations as Training Data for End-to-End Neural Diarization","date":"2022-11-12","arxiv_id":"2211.06750","repositories_listed":3,"syntology":null},{"url":"/paper/road-the-road-event-awareness-dataset-for","title":"ROAD: The ROad event Awareness Dataset for Autonomous Driving","date":"2021-02-23","arxiv_id":"2102.11585","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/pyannoteaudio-neural-building-blocks-for","title":"pyannote.audio: neural building blocks for speaker diarization","date":"2019-11-04","arxiv_id":"1911.01255","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/rvad-an-unsupervised-segment-based-robust","title":"rVAD: An Unsupervised Segment-Based Robust Voice Activity Detection Method","date":"2019-06-09","arxiv_id":"1906.03588","repositories_listed":3,"syntology":null},{"url":"/paper/fine-grained-activity-recognition-in-baseball","title":"Fine-grained Activity Recognition in Baseball Videos","date":"2018-04-09","arxiv_id":"1804.03247","repositories_listed":3,"syntology":null},{"url":"/paper/r-c3d-region-convolutional-3d-network-for","title":"R-C3D: Region Convolutional 3D Network for Temporal Activity Detection","date":"2017-03-22","arxiv_id":"1703.07814","repositories_listed":3,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/an-end-to-end-architecture-for-keyword","title":"An End-to-End Architecture for Keyword Spotting and Voice Activity Detection","date":"2016-11-28","arxiv_id":"1611.09405","repositories_listed":3,"syntology":null},{"url":"/paper/temporal-activity-detection-in-untrimmed","title":"Temporal Activity Detection in Untrimmed Videos with Recurrent Neural Networks","date":"2016-08-29","arxiv_id":"1608.08128","repositories_listed":3,"syntology":null},{"url":"/paper/ivrit-ai-a-comprehensive-dataset-of-hebrew","title":"ivrit.ai: A Comprehensive Dataset of Hebrew Speech for AI Research and Development","date":"2023-07-17","arxiv_id":"2307.08720","repositories_listed":2,"syntology":null},{"url":"/paper/speaker-embedding-aware-neural-diarization","title":"Speaker Embedding-aware Neural Diarization for Flexible Number of Speakers with Textual Information","date":"2021-11-28","arxiv_id":"2111.13694","repositories_listed":2,"syntology":null},{"url":"/paper/bertraffic-a-robust-bert-based-approach-for","title":"BERTraffic: BERT-based Joint Speaker Role and Speaker Change Detection for Air Traffic Control Communications","date":"2021-10-12","arxiv_id":"2110.05781","repositories_listed":2,"syntology":null},{"url":"/paper/end-to-end-speaker-segmentation-for-overlap","title":"End-to-end speaker segmentation for overlap-aware resegmentation","date":"2021-04-08","arxiv_id":"2104.04045","repositories_listed":2,"syntology":null},{"url":"/paper/voxlingua107-a-dataset-for-spoken-language-1","title":"VoxLingua107: a Dataset for Spoken Language Recognition","date":"2020-11-25","arxiv_id":"2011.12998","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/harvesting-ambient-rf-for-presence-detection","title":"Harvesting Ambient RF for Presence Detection Through Deep Learning","date":"2020-02-13","arxiv_id":"2002.05770","repositories_listed":2,"syntology":null},{"url":"/paper/personal-vad-speaker-conditioned-voice","title":"Personal VAD: Speaker-Conditioned Voice Activity Detection","date":"2019-08-12","arxiv_id":"1908.04284","repositories_listed":2,"syntology":null},{"url":"/paper/learning-latent-super-events-to-detect","title":"Learning Latent Super-Events to Detect Multiple Activities in Videos","date":"2017-12-05","arxiv_id":"1712.01938","repositories_listed":2,"syntology":null},{"url":"/paper/speaker-diarization-with-overlapping","title":"Speaker Diarization with Overlapping Community Detection Using Graph Attention Networks and Label Propagation Algorithm","date":"2025-06-03","arxiv_id":"2506.02610","repositories_listed":1,"syntology":null},{"url":"/paper/2505-10879","title":"Multi-Stage Speaker Diarization for Noisy Classrooms","date":"2025-05-16","arxiv_id":"2505.10879","repositories_listed":1,"syntology":null},{"url":"/paper/optimizing-large-language-models-for-esg","title":"Optimizing Large Language Models for ESG Activity Detection in Financial Texts","date":"2025-02-28","arxiv_id":"2502.21112","repositories_listed":1,"syntology":null},{"url":"/paper/vanpy-voice-analysis-framework","title":"VANPY: Voice Analysis Framework","date":"2025-02-17","arxiv_id":"2502.17579","repositories_listed":1,"syntology":null},{"url":"/paper/when-do-they-stop-a-first-step-towards","title":"When do they StOP?: A First Step Towards Automatically Identifying Team Communication in the Operating Room","date":"2025-02-12","arxiv_id":"2502.08299","repositories_listed":1,"syntology":null},{"url":"/paper/pre-equalization-aided-grant-free-massive","title":"Pre-Equalization Aided Grant-Free Massive Access in Massive MIMO System","date":"2025-02-10","arxiv_id":"2502.06239","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-detection-and-prediction-of-namd","title":"Automatic detection and prediction of nAMD activity change in retinal OCT using Siamese networks and Wasserstein Distance for ordinality","date":"2025-01-24","arxiv_id":"2501.14323","repositories_listed":1,"syntology":null},{"url":"/paper/wifi-csi-based-temporal-activity-detection","title":"WiFi CSI Based Temporal Activity Detection via Dual Pyramid Network","date":"2024-12-19","arxiv_id":"2412.16233","repositories_listed":1,"syntology":null},{"url":"/paper/automating-feedback-analysis-in-surgical","title":"Automating Feedback Analysis in Surgical Training: Detection, Categorization, and Assessment","date":"2024-12-01","arxiv_id":"2412.00760","repositories_listed":1,"syntology":null},{"url":"/paper/a-framework-for-adapting-human-robot","title":"A Framework for Adapting Human-Robot Interaction to Diverse User Groups","date":"2024-10-15","arxiv_id":"2410.11377","repositories_listed":1,"syntology":null},{"url":"/paper/inagvad-a-challenging-french-tv-and-radio","title":"InaGVAD : a Challenging French TV and Radio Corpus Annotated for Speech Activity Detection and Speaker Gender Segmentation","date":"2024-06-06","arxiv_id":"2406.04429","repositories_listed":1,"syntology":null},{"url":"/paper/activity-detection-for-massive-connectivity","title":"Activity Detection for Massive Connectivity in Cell-free Networks with Unknown Large-scale Fading, Channel Statistics, Noise Variance, and Activity Probability: A Bayesian Approach","date":"2024-01-30","arxiv_id":"2401.16775","repositories_listed":1,"syntology":null},{"url":"/paper/online-speaker-diarization-of-meetings-guided","title":"Online speaker diarization of meetings guided by speech separation","date":"2024-01-30","arxiv_id":"2402.00067","repositories_listed":1,"syntology":null}],"syntology_records":5,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}