{"url":"/dataset/activitynet","name":"ActivityNet","full_name":null,"description_markdown":"The **ActivityNet** dataset contains 200 different types of activities and a total of 849 hours of videos collected from YouTube. ActivityNet is the largest benchmark for temporal activity detection to date in terms of both the number of activity categories and number of videos, making the task particularly challenging. Version 1.3 of the dataset contains 19994 untrimmed videos in total and is divided into three disjoint subsets, training, validation, and testing by a ratio of 2:1:1. On average, each activity category has 137 untrimmed videos. Each video on average has 1.41 activities which are annotated with temporal boundaries. The ground-truth annotations of test videos are not public.\r\n\r\nSource: [Dynamic Temporal Pyramid Network: A Closer Look at Multi-Scale Modeling for Activity Detection](https://arxiv.org/abs/1808.02536)","description_withheld":null,"homepage":"http://activity-net.org/","introduced_date":"2015-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/activitynet-a-large-scale-video-benchmark-for","title":"ActivityNet: A Large-Scale Video Benchmark for Human Activity Understanding","first_author":"Fabian Caba Heilbron","url":null},"license":{"name":"Unknown","url":null},"modalities":[{"name":"Videos","url":"/datasets/modality/videos"}],"tasks":[{"name":"Action Recognition","url":"/task/action-recognition-in-videos","datasets_with_task":"/datasets/task/action-recognition-in-videos"},{"name":"Visual Question Answering (VQA)","url":"/task/visual-question-answering","datasets_with_task":"/datasets/task/visual-question-answering"},{"name":"Temporal Action Localization","url":"/task/action-recognition","datasets_with_task":"/datasets/task/action-recognition"},{"name":"Action Detection","url":"/task/action-detection","datasets_with_task":"/datasets/task/action-detection"},{"name":"Video Retrieval","url":"/task/video-retrieval","datasets_with_task":"/datasets/task/video-retrieval"},{"name":"Action Classification","url":"/task/action-classification","datasets_with_task":"/datasets/task/action-classification"},{"name":"Action Recognition In Videos","url":"/task/action-recognition-in-videos-2","datasets_with_task":"/datasets/task/action-recognition-in-videos-2"},{"name":"Weakly Supervised Action Localization","url":"/task/weakly-supervised-action-localization","datasets_with_task":"/datasets/task/weakly-supervised-action-localization"},{"name":"Zero-Shot Video Retrieval","url":"/task/zero-shot-video-retrieval","datasets_with_task":"/datasets/task/zero-shot-video-retrieval"},{"name":"Zero-Shot Action Recognition","url":"/task/zero-shot-action-recognition","datasets_with_task":"/datasets/task/zero-shot-action-recognition"},{"name":"GZSL Video Classification","url":"/task/gzsl-video-classification","datasets_with_task":"/datasets/task/gzsl-video-classification"},{"name":"Temporal Action Proposal Generation","url":"/task/temporal-action-proposal-generation","datasets_with_task":"/datasets/task/temporal-action-proposal-generation"},{"name":"Weakly-supervised Temporal Action Localization","url":"/task/weakly-supervised-temporal-action","datasets_with_task":"/datasets/task/weakly-supervised-temporal-action"},{"name":"ZSL Video Classification","url":"/task/zsl-video-classification","datasets_with_task":"/datasets/task/zsl-video-classification"},{"name":"Few Shot Temporal Action Localization","url":"/task/few-shot-temporal-action-localization","datasets_with_task":"/datasets/task/few-shot-temporal-action-localization"},{"name":"Semi-Supervised Action Detection","url":"/task/semi-supervised-action-detection","datasets_with_task":"/datasets/task/semi-supervised-action-detection"},{"name":"Zero-Shot Action Detection","url":"/task/zero-shot-action-detection","datasets_with_task":"/datasets/task/zero-shot-action-detection"}],"languages":[],"variants":["ActivityNet","ActivityNet-1.3","ActivityNet-1.2","ActivityNet-GZSL (cls)","ActivityNet-GZSL(main)"],"data_loaders":[{"repo":"https://github.com/open-mmlab/mmaction2","url":"https://github.com/open-mmlab/mmaction2/blob/master/tools/data/activitynet/README.md","frameworks":["pytorch"]}],"num_papers_in_archive":807,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/temporal-action-localization-on-activitynet","task":"Temporal Action Localization","dataset_variant":"ActivityNet-1.3","rows":33,"metrics":["mAP","mAP IOU@0.5","mAP IOU@0.75","mAP IOU@0.95"],"first_row_in_archive_order":{"model":"RDFA-S6 (InternVideo2-6B)","paper":"/paper/enhancing-temporal-action-localization","metrics":{"mAP":"42.9","mAP IOU@0.5":"64.1","mAP IOU@0.75":"44.0","mAP IOU@0.95":"10.6"},"code_links":[{"title":"lsy0882/RDFA-S6","url":"https://github.com/lsy0882/RDFA-S6"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/video-retrieval-on-activitynet","task":"Video Retrieval","dataset_variant":"ActivityNet","rows":31,"metrics":["text-to-video R@1","text-to-video R@5","text-to-video R@10","text-to-video R@50","text-to-video Mean Rank","text-to-video Median Rank","video-to-text R@1","video-to-text R@5","video-to-text Mean Rank","video-to-text Median Rank","video-to-text R@10","video-to-text R@50"],"first_row_in_archive_order":{"model":"InternVideo2-6B","paper":"/paper/internvideo2-scaling-video-foundation-models","metrics":{"text-to-video R@1":"74.1","video-to-text R@1":"69.7"},"code_links":[{"title":"opengvlab/internvideo","url":"https://github.com/opengvlab/internvideo"},{"title":"opengvlab/internvideo2","url":"https://github.com/opengvlab/internvideo2"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/weakly-supervised-action-localization-on-2","task":"Weakly Supervised Action Localization","dataset_variant":"ActivityNet-1.2","rows":19,"metrics":["Mean mAP","mAP@0.5"],"first_row_in_archive_order":{"model":"SAL","paper":"/paper/multilevel-semantic-and-adaptive-actionness","metrics":{"Mean mAP":"30.8","mAP@0.5":"48.5"},"code_links":[{"title":"lizhilin-ustc/SAL","url":"https://github.com/lizhilin-ustc/SAL"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/weakly-supervised-action-localization-on-1","task":"Weakly Supervised Action Localization","dataset_variant":"ActivityNet-1.3","rows":17,"metrics":["mAP@0.5:0.95","mAP@0.5"],"first_row_in_archive_order":{"model":"SAL","paper":"/paper/multilevel-semantic-and-adaptive-actionness","metrics":{"mAP@0.5":"44.5","mAP@0.5:0.95":"28.8"},"code_links":[{"title":"lizhilin-ustc/SAL","url":"https://github.com/lizhilin-ustc/SAL"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/action-recognition-in-videos-on-activitynet","task":"Action Recognition","dataset_variant":"ActivityNet","rows":16,"metrics":["mAP"],"first_row_in_archive_order":{"model":"Text4Vis (w/ ViT-L)","paper":"/paper/transferring-textual-knowledge-for-visual","metrics":{"mAP":"96.9"},"code_links":[{"title":"whwu95/Cap4Video","url":"https://github.com/whwu95/Cap4Video"},{"title":"whwu95/text4vis","url":"https://github.com/whwu95/text4vis"},{"title":"whwu95/GPT4Vis","url":"https://github.com/whwu95/GPT4Vis"},{"title":"whwu95/BIKE","url":"https://github.com/whwu95/BIKE"},{"title":"whwu95/ATM","url":"https://github.com/whwu95/ATM"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/zero-shot-video-retrieval-on-activitynet","task":"Zero-Shot Video Retrieval","dataset_variant":"ActivityNet","rows":12,"metrics":["text-to-video R@1","text-to-video R@5","text-to-video R@10","video-to-text R@1","video-to-text R@5","video-to-text R@10"],"first_row_in_archive_order":{"model":"InternVideo2-6B","paper":"/paper/internvideo2-scaling-video-foundation-models","metrics":{"text-to-video R@1":"63.2","text-to-video R@10":"92.5","text-to-video R@5":"85.6","video-to-text R@1":"56.5","video-to-text R@10":"90.3","video-to-text R@5":"82.8"},"code_links":[{"title":"opengvlab/internvideo","url":"https://github.com/opengvlab/internvideo"},{"title":"opengvlab/internvideo2","url":"https://github.com/opengvlab/internvideo2"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/temporal-action-proposal-generation-on","task":"Temporal Action Proposal Generation","dataset_variant":"ActivityNet-1.3","rows":11,"metrics":["AR@100","AUC (val)","AUC (test)"],"first_row_in_archive_order":{"model":"AOE-Net","paper":"/paper/aoe-net-entities-interactions-modeling-with","metrics":{"AR@100":"77.67","AUC (test)":"70.10","AUC (val)":"69.71"},"code_links":[{"title":"uark-aicv/aoe-net","url":"https://github.com/uark-aicv/aoe-net"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/gzsl-video-classification-on-activitynet-gzsl-1","task":"GZSL Video Classification","dataset_variant":"ActivityNet-GZSL(main)","rows":7,"metrics":["HM","ZSL"],"first_row_in_archive_order":{"model":"KDA","paper":"/paper/boosting-audio-visual-zero-shot-learning-with","metrics":{"HM":"19.67","ZSL":"14.00"},"code_links":[{"title":"chenhaoxing/KDA","url":"https://github.com/chenhaoxing/KDA"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/zero-shot-action-recognition-on-activitynet","task":"Zero-Shot Action Recognition","dataset_variant":"ActivityNet","rows":5,"metrics":["Top-1 Accuracy"],"first_row_in_archive_order":{"model":"BIKE","paper":"/paper/bidirectional-cross-modal-knowledge","metrics":{"Top-1 Accuracy":"86.2"},"code_links":[{"title":"whwu95/Cap4Video","url":"https://github.com/whwu95/Cap4Video"},{"title":"whwu95/text4vis","url":"https://github.com/whwu95/text4vis"},{"title":"whwu95/GPT4Vis","url":"https://github.com/whwu95/GPT4Vis"},{"title":"whwu95/BIKE","url":"https://github.com/whwu95/BIKE"},{"title":"whwu95/ATM","url":"https://github.com/whwu95/ATM"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/gzsl-video-classification-on-activitynet-gzsl","task":"GZSL Video Classification","dataset_variant":"ActivityNet-GZSL (cls)","rows":4,"metrics":["HM","ZSL"],"first_row_in_archive_order":{"model":"KDA","paper":"/paper/boosting-audio-visual-zero-shot-learning-with","metrics":{"HM":"17.95","ZSL":"11.85"},"code_links":[{"title":"chenhaoxing/KDA","url":"https://github.com/chenhaoxing/KDA"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/action-classification-on-activitynet-12","task":"Action Classification","dataset_variant":"ActivityNet-1.2","rows":3,"metrics":["mAP"],"first_row_in_archive_order":{"model":"W-TALC","paper":"/paper/w-talc-weakly-supervised-temporal-activity","metrics":{"mAP":"93.2"},"code_links":[{"title":"sujoyp/wtalc-pytorch","url":"https://github.com/sujoyp/wtalc-pytorch"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/action-classification-on-activitynet","task":"Action Classification","dataset_variant":"ActivityNet","rows":1,"metrics":["Top 1 Accuracy","Top 5 Accuracy"],"first_row_in_archive_order":{"model":"UniFormerV2-L","paper":"/paper/uniformerv2-spatiotemporal-learning-by-arming","metrics":{"Top 1 Accuracy":"94.7","Top 5 Accuracy":"99.5"},"code_links":[{"title":"OpenGVLab/UniFormerV2","url":"https://github.com/OpenGVLab/UniFormerV2"},{"title":"innat/UniFormerV2","url":"https://github.com/innat/UniFormerV2"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/action-recognition-in-videos-on-activitynet-1","task":"Action Recognition In Videos","dataset_variant":"ActivityNet","rows":1,"metrics":["mAP"],"first_row_in_archive_order":{"model":"LSTM + Pretrained on YT-8M","paper":"/paper/youtube-8m-a-large-scale-video-classification","metrics":{"mAP":"75.6"},"code_links":[{"title":"google/youtube-8m","url":"https://github.com/google/youtube-8m"},{"title":"AKASH2907/Content-based-Video-Recommendation","url":"https://github.com/AKASH2907/Content-based-Video-Recommendation"},{"title":"AKASH2907/Content-based-Video-Relevance-Prediction","url":"https://github.com/AKASH2907/Content-based-Video-Relevance-Prediction"},{"title":"taufikxu/youtube","url":"https://github.com/taufikxu/youtube"},{"title":"Cloud-Computing-IoT/speakEasy","url":"https://github.com/Cloud-Computing-IoT/speakEasy"},{"title":"Cloud-Computing-IoT/Cloud-Enabled-Smart-Speaker","url":"https://github.com/Cloud-Computing-IoT/Cloud-Enabled-Smart-Speaker"},{"title":"boseaslcohort/youtube-8m","url":"https://github.com/boseaslcohort/youtube-8m"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/few-shot-temporal-action-localization-on","task":"Few Shot Temporal Action Localization","dataset_variant":"ActivityNet","rows":1,"metrics":["mIoU"],"first_row_in_archive_order":{"model":"FS-QAT","paper":"/paper/few-shot-temporal-action-localization-with","metrics":{"mIoU":"38.5"},"code_links":[{"title":"sauradip/fewshotQAT","url":"https://github.com/sauradip/fewshotQAT"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/temporal-action-localization-on-activitynet-1","task":"Temporal Action Localization","dataset_variant":"ActivityNet-1.2","rows":1,"metrics":["mAP IOU@0.5","mAP IOU@0.1","mAP IOU@0.3","mAP IOU@0.7"],"first_row_in_archive_order":{"model":"DeepMetricLearner","paper":"/paper/weakly-supervised-temporal-action-1","metrics":{"mAP IOU@0.1":"60.5","mAP IOU@0.3":"48.4","mAP IOU@0.5":"35.2","mAP IOU@0.7":"16.3"},"code_links":[{"title":"asrafulashiq/wsad","url":"https://github.com/asrafulashiq/wsad"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-question-answering-vqa-on-activitynet-1","task":"Visual Question Answering (VQA)","dataset_variant":"ActivityNet","rows":1,"metrics":["ClipMatch@1","ClipMatch@5","Contains","ExactMatch","Follow-up ClipMatch@1","Follow-up ClipMatch@5","Follow-up Contains","Follow-up ExactMatch"],"first_row_in_archive_order":{"model":"BLIP-2 T5","paper":"/paper/open-ended-vqa-benchmarking-of-vision","metrics":{"ClipMatch@1":"53.39","ClipMatch@5":"74.71","Contains":"15.70","ExactMatch":"7.07","Follow-up ClipMatch@1":"62.02","Follow-up ClipMatch@5":"75.13","Follow-up Contains":"18.09","Follow-up ExactMatch":"8.84"},"code_links":[{"title":"lmb-freiburg/ovqa","url":"https://github.com/lmb-freiburg/ovqa"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/weakly-supervised-temporal-action-1","task":"Weakly-supervised Temporal Action Localization","dataset_variant":"ActivityNet-1.3","rows":1,"metrics":["mAP","mAP IOU@0.5","mAP IOU@0.75","mAP IOU@0.95"],"first_row_in_archive_order":{"model":"ASM-Loc","paper":"/paper/asm-loc-action-aware-segment-modeling-for","metrics":{"mAP":"25.1","mAP IOU@0.5":"41.0","mAP IOU@0.75":"24.9","mAP IOU@0.95":"6.2"},"code_links":[{"title":"boheumd/asm-loc","url":"https://github.com/boheumd/asm-loc"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/gramian-multimodal-representation-learning","title":"Gramian Multimodal Representation Learning and Alignment","date":"2024-12-16","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":12,"samples_ran":2,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/locate-gat-modeling-multi-scale-local-context","title":"LoCATe-GAT: Modeling Multi-Scale Local Context and Action Relationships for Zero-Shot Action Recognition","date":"2024-11-27","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/multilevel-semantic-and-adaptive-actionness","title":"Multilevel semantic and adaptive actionness learning for weakly supervised temporal action localization","date":"2024-11-24","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/enhancing-temporal-action-localization","title":"Enhancing Temporal Action Localization: Advanced S6 Modeling with Recurrent Mechanism","date":"2024-07-18","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/weakly-supervised-temporal-action-8","title":"Weakly supervised temporal action localization with actionness-guided false positive suppression","date":"2024-04-15","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/unimd-towards-unifying-moment-retrieval-and","title":"UniMD: Towards Unifying Moment Retrieval and Temporal Action Detection","date":"2024-04-07","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/internvideo2-scaling-video-foundation-models","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","date":"2024-03-22","rows_on_this_dataset":6,"code_links":2,"syntology":null},{"paper":"/paper/vid-tldr-training-free-token-merging-for","title":"vid-TLDR: Training Free Token merging for Light-weight Video Transformer","date":"2024-03-20","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/video-mamba-suite-state-space-model-as-a","title":"Video Mamba Suite: State Space Model as a Versatile Alternative for Video Understanding","date":"2024-03-14","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":13,"samples_ran":8,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/open-ended-vqa-benchmarking-of-vision","title":"Open-ended VQA benchmarking of Vision-Language models by exploiting Classification datasets and their semantic hierarchy","date":"2024-02-11","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/revisiting-foreground-and-background-1","title":"Revisiting Foreground and Background Separation in Weakly-supervised Temporal Action Localization: A Clustering-based Approach","date":"2023-12-21","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/rtq-rethinking-video-language-understanding","title":"RTQ: Rethinking Video-language Understanding Based on Image-text Model","date":"2023-12-01","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/end-to-end-temporal-action-detection-with-1b","title":"End-to-End Temporal Action Detection with 1B Parameters Across 1000 Frames","date":"2023-11-28","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/boosting-audio-visual-zero-shot-learning-with","title":"Boosting Audio-visual Zero-shot Learning with Large Language Models","date":"2023-11-21","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/testa-temporal-spatial-token-aggregation-for","title":"TESTA: Temporal-Spatial Token Aggregation for Long-form Video-Language Understanding","date":"2023-10-29","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":15,"samples_ran":11,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/languagebind-extending-video-language","title":"LanguageBind: Extending Video-Language Pretraining to N-modality by Language-based Semantic Alignment","date":"2023-10-03","rows_on_this_dataset":2,"code_links":6,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":14,"samples_ran":9,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/one-for-all-video-conversation-is-feasible","title":"BT-Adapter: Video Conversation is Feasible Without Video Instruction Tuning","date":"2023-09-27","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":4,"samples_ran":1,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/dual-modal-attention-enhanced-text-video","title":"Dual-Modal Attention-Enhanced Text-Video Retrieval with Triplet Partial Margin Contrastive Learning","date":"2023-09-20","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/hyperbolic-audio-visual-zero-shot-learning","title":"Hyperbolic Audio-visual Zero-shot Learning","date":"2023-08-24","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/unloc-a-unified-framework-for-video","title":"UnLoc: A Unified Framework for Video Localization Tasks","date":"2023-08-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/actionness-inconsistency-guided-contrastive","title":"Actionness Inconsistency-guided Contrastive Learning for Weakly-supervised Temporal Action Localization","date":"2023-06-26","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/cosa-concatenated-sample-pretrained-vision","title":"COSA: Concatenated Sample Pretrained Vision-Language Foundation Model","date":"2023-06-15","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/vast-a-vision-audio-subtitle-text-omni-1","title":"VAST: A Vision-Audio-Subtitle-Text Omni-Modality Foundation Model and Dataset","date":"2023-05-29","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":42,"samples_ran":15,"samples_unverified":27,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/proposal-based-multiple-instance-learning-for-1","title":"Proposal-Based Multiple Instance Learning for Weakly-Supervised Temporal Action Localization","date":"2023-05-29","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/valor-vision-audio-language-omni-perception","title":"VALOR: Vision-Audio-Language Omni-Perception Pretraining Model and Dataset","date":"2023-04-17","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/improve-temporal-action-proposals-using","title":"Improve Temporal Action Proposals using Hierarchical Context","date":"2023-04-03","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/unmasked-teacher-towards-training-efficient","title":"Unmasked Teacher: Towards Training-Efficient Video Foundation Models","date":"2023-03-28","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":8,"samples_ran":5,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/video-text-as-game-players-hierarchical","title":"Video-Text as Game Players: Hierarchical Banzhaf Interaction for Cross-Modal Representation Learning","date":"2023-03-25","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":16,"samples_ran":12,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/diffusionret-generative-text-video-retrieval","title":"DiffusionRet: Generative Text-Video Retrieval with Diffusion Model","date":"2023-03-17","rows_on_this_dataset":2,"code_links":4,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":6,"samples_ran":5,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/tridet-temporal-action-detection-with","title":"TriDet: Temporal Action Detection with Relative Boundary Modeling","date":"2023-03-13","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":15,"samples_ran":5,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/two-stream-networks-for-weakly-supervised","title":"Two-Stream Networks for Weakly-Supervised Temporal Action Localization With Semantic-Aware Mechanisms","date":"2023-01-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/pivotal-prior-driven-supervision-for-weakly","title":"PivoTAL: Prior-Driven Supervision for Weakly-Supervised Temporal Action Localization","date":"2023-01-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/bidirectional-cross-modal-knowledge","title":"Bidirectional Cross-Modal Knowledge Exploration for Video Recognition with Pre-trained Vision-Language Models","date":"2022-12-31","rows_on_this_dataset":2,"code_links":5,"syntology":null},{"paper":"/paper/hitea-hierarchical-temporal-aware-video","title":"HiTeA: Hierarchical Temporal-Aware Video-Language Pre-training","date":"2022-12-30","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/vindlu-a-recipe-for-effective-video-and","title":"VindLU: A Recipe for Effective Video-and-Language Pretraining","date":"2022-12-09","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/video-text-modeling-with-zero-shot-transfer","title":"VideoCoCa: Video-Text Modeling with Zero-Shot Transfer from Contrastive Captioners","date":"2022-12-09","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/internvideo-general-video-foundation-models","title":"InternVideo: General Video Foundation Models via Generative and Discriminative Learning","date":"2022-12-06","rows_on_this_dataset":3,"code_links":2,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/expectation-maximization-contrastive-learning","title":"Expectation-Maximization Contrastive Learning for Compact Video-and-Language Representations","date":"2022-11-21","rows_on_this_dataset":2,"code_links":4,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/aoe-net-entities-interactions-modeling-with","title":"AOE-Net: Entities Interactions Modeling with Adaptive Attention Mechanism for Temporal Action Proposals Generation","date":"2022-10-05","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/uniformerv2-spatiotemporal-learning-by-arming","title":"UniFormerV2: Spatiotemporal Learning by Arming Image ViTs with Video UniFormer","date":"2022-09-22","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/clip-vip-adapting-pre-trained-image-text","title":"CLIP-ViP: Adapting Pre-trained Image-Text Model to Video-Language Representation Alignment","date":"2022-09-14","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":4,"samples_ran":4,"samples_unverified":0,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/temporal-saliency-query-network-for-efficient","title":"Temporal Saliency Query Network for Efficient Video Recognition","date":"2022-07-21","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/nsnet-non-saliency-suppression-sampler-for","title":"NSNet: Non-saliency Suppression Sampler for Efficient Video Recognition","date":"2022-07-21","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/temporal-and-cross-modal-attention-for-audio","title":"Temporal and cross-modal attention for audio-visual zero-shot learning","date":"2022-07-20","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/x-clip-end-to-end-multi-grained-contrastive","title":"X-CLIP: End-to-End Multi-grained Contrastive Learning for Video-Text Retrieval","date":"2022-07-15","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/temporal-action-detection-with-global","title":"Proposal-Free Temporal Action Detection via Global Segmentation Mask Learning","date":"2022-07-14","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/transferring-textual-knowledge-for-visual","title":"Revisiting Classifier: Transferring Vision-Language Models for Video Recognition","date":"2022-07-04","rows_on_this_dataset":2,"code_links":5,"syntology":null},{"paper":"/paper/weakly-supervised-action-localization-via","title":"Weakly-Supervised Temporal Action Localization by Progressive Complementary Learning","date":"2022-06-22","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/revealing-single-frame-bias-for-video-and","title":"Revealing Single Frame Bias for Video-and-Language Learning","date":"2022-06-07","rows_on_this_dataset":3,"code_links":2,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":12,"samples_ran":9,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/cross-modal-representation-learning-for-zero","title":"Cross-modal Representation Learning for Zero-shot Action Recognition","date":"2022-05-03","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/centerclip-token-clustering-for-efficient","title":"CenterCLIP: Token Clustering for Efficient Text-Video Retrieval","date":"2022-05-02","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/hunyuan-tvr-for-text-video-retrivial","title":"Tencent Text-Video Retrieval: Hierarchical Cross-Modal Interactions with Multi-Level Representations","date":"2022-04-07","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/an-empirical-study-of-end-to-end-temporal","title":"An Empirical Study of End-to-End Temporal Action Detection","date":"2022-04-06","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":11,"samples_ran":10,"samples_unverified":1,"pointer_only_for_licence":11,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/attribute-prototype-network-for-any-shot","title":"Attribute Prototype Network for Any-Shot Learning","date":"2022-04-04","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/asm-loc-action-aware-segment-modeling-for","title":"ASM-Loc: Action-aware Segment Modeling for Weakly-Supervised Temporal Action Localization","date":"2022-03-29","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/abn-agent-aware-boundary-networks-for","title":"ABN: Agent-Aware Boundary Networks for Temporal Action Proposal Generation","date":"2022-03-16","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/audio-visual-generalised-zero-shot-learning","title":"Audio-visual Generalised Zero-shot Learning with Cross-modal Attention and Language","date":"2022-03-07","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/actionformer-localizing-moments-of-actions","title":"ActionFormer: Localizing Moments of Actions with Transformers","date":"2022-02-16","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/dcan-improving-temporal-action-detection-via","title":"DCAN: Improving Temporal Action Detection via Dual Context Aggregation","date":"2021-12-07","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/low-fidelity-video-encoder-optimization-for","title":"Low-Fidelity Video Encoder Optimization for Temporal Action Localization","date":"2021-12-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/graph-convolutional-module-for-temporal","title":"Graph Convolutional Module for Temporal Action Localization in Videos","date":"2021-12-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/advancing-high-resolution-video-language","title":"Advancing High-Resolution Video-Language Representation with Large-Scale Video Transcriptions","date":"2021-11-19","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/video-and-text-matching-with-conditioned","title":"Video and Text Matching with Conditioned Embeddings","date":"2021-10-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/aei-actors-environment-interaction-with","title":"AEI: Actors-Environment Interaction with Adaptive Attention for Temporal Action Proposals Generation","date":"2021-10-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/few-shot-temporal-action-localization-with","title":"Few-Shot Temporal Action Localization with Query Adaptive Transformer","date":"2021-10-20","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/improving-video-text-retrieval-by-multi","title":"Improving Video-Text Retrieval by Multi-Stream Corpus Alignment and Dual Softmax Loss","date":"2021-09-09","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/taco-token-aware-cascade-contrastive-learning","title":"TACo: Token-aware Cascade Contrastive Learning for Video-Text Alignment","date":"2021-08-23","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/learning-action-completeness-from-points-for","title":"Learning Action Completeness from Points for Weakly-supervised Temporal Action Localization","date":"2021-08-11","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/hear-me-out-fusional-approaches-for-audio","title":"Hear Me Out: Fusional Approaches for Audio Augmented Temporal Action Localization","date":"2021-06-27","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/proposal-relation-network-for-temporal-action","title":"Proposal Relation Network for Temporal Action Detection","date":"2021-06-20","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/end-to-end-temporal-action-detection-with","title":"End-to-end Temporal Action Detection with Transformer","date":"2021-06-18","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/dsanet-dynamic-segment-aggregation-network","title":"DSANet: Dynamic Segment Aggregation Network for Video-Level Representation Learning","date":"2021-05-25","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":13,"samples_ran":0,"samples_unverified":13,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/weakly-supervised-action-selection-learning","title":"Weakly Supervised Action Selection Learning in Video","date":"2021-05-06","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/clip4clip-an-empirical-study-of-clip-for-end","title":"CLIP4Clip: An Empirical Study of CLIP for End to End Video Clip Retrieval","date":"2021-04-18","rows_on_this_dataset":1,"code_links":5,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/self-supervised-learning-for-semi-supervised","title":"Self-Supervised Learning for Semi-Supervised Temporal Action Proposal","date":"2021-04-07","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/acm-net-action-context-modeling-network-for","title":"ACM-Net: Action Context Modeling Network for Weakly-Supervised Temporal Action Localization","date":"2021-04-07","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/cola-weakly-supervised-temporal-action","title":"CoLA: Weakly-Supervised Temporal Action Localization with Snippet Contrastive Learning","date":"2021-03-30","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/temporal-context-aggregation-network-for","title":"Temporal Context Aggregation Network for Temporal Action Proposal Refinement","date":"2021-03-24","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/2d-or-not-2d-adaptive-3d-convolution","title":"2D or not 2D? Adaptive 3D Convolution Selection for Efficient Video Recognition","date":"2020-12-29","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/smart-frame-selection-for-action-recognition","title":"SMART Frame Selection for Action Recognition","date":"2020-12-19","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/d2-net-weakly-supervised-action-localization","title":"D2-Net: Weakly-Supervised Action Localization via Discriminative Embeddings and Denoised Activations","date":"2020-12-11","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/video-self-stitching-graph-network-for","title":"Video Self-Stitching Graph Network for Temporal Action Localization","date":"2020-11-30","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/tsp-temporally-sensitive-pretraining-of-video","title":"TSP: Temporally-Sensitive Pretraining of Video Encoders for Localization Tasks","date":"2020-11-23","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/boundary-sensitive-pre-training-for-temporal","title":"Boundary-sensitive Pre-training for Temporal Localization in Videos","date":"2020-11-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/bsn-complementary-boundary-regressor-with","title":"BSN++: Complementary Boundary Regressor with Scale-Balanced Relation Modeling for Temporal Action Proposal Generation","date":"2020-09-15","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":9,"samples_ran":7,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/boundary-content-graph-neural-network-for","title":"Boundary Content Graph Neural Network for Temporal Action Proposal Generation","date":"2020-08-04","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/multi-modal-transformer-for-video-retrieval","title":"Multi-modal Transformer for Video Retrieval","date":"2020-07-21","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":7,"samples_ran":7,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/adversarial-background-aware-loss-for-weakly","title":"Adversarial Background-Aware Loss for Weakly-supervised Temporal Activity Localization","date":"2020-07-13","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/dynamic-sampling-networks-for-efficient","title":"Dynamic Sampling Networks for Efficient Action Recognition in Videos","date":"2020-06-28","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/background-modeling-via-uncertainty","title":"Weakly-supervised Temporal Action Localization by Uncertainty Modeling","date":"2020-06-12","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/avgzslnet-audio-visual-generalized-zero-shot","title":"AVGZSLNet: Audio-Visual Generalized Zero-Shot Learning by Reconstructing Label Features from Multi-Modal Embeddings","date":"2020-05-27","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/weakly-supervised-action-localization-by-2","title":"Weakly-Supervised Action Localization by Generative Attention Modeling","date":"2020-03-27","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/sf-net-single-frame-supervision-for-temporal","title":"SF-Net: Single-Frame Supervision for Temporal Action Localization","date":"2020-03-15","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/rethinking-zero-shot-video-classification-end","title":"Rethinking Zero-shot Video Classification: End-to-end Training for Realistic Applications","date":"2020-03-03","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":12,"samples_ran":7,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/weakly-supervised-temporal-action-1","title":"Weakly Supervised Temporal Action Localization Using Deep Metric Learning","date":"2020-01-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/listen-to-look-action-recognition-by","title":"Listen to Look: Action Recognition by Previewing Audio","date":"2019-12-10","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/g-tad-sub-graph-localization-for-temporal","title":"G-TAD: Sub-Graph Localization for Temporal Action Detection","date":"2019-11-26","rows_on_this_dataset":1,"code_links":7,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":18,"samples_ran":6,"samples_unverified":12,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/background-suppression-network-for-weakly","title":"Background Suppression Network for Weakly-supervised Temporal Action Localization","date":"2019-11-22","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/coordinated-joint-multimodal-embeddings-for","title":"Coordinated Joint Multimodal Embeddings for Generalized Audio-Visual Zeroshot Classification and Retrieval of Videos","date":"2019-10-19","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/weakly-supervised-temporal-action","title":"Weakly Supervised Temporal Action Localization Through Contrast Based Evaluation Networks","date":"2019-10-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/temporal-structure-mining-for-weakly","title":"Temporal Structure Mining for Weakly Supervised Action Detection","date":"2019-10-01","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/graph-convolutional-networks-for-temporal","title":"Graph Convolutional Networks for Temporal Action Localization","date":"2019-09-07","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/3c-net-category-count-and-center-loss-for","title":"3C-Net: Category Count and Center Loss for Weakly-Supervised Action Localization","date":"2019-08-22","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/use-what-you-have-video-retrieval-using","title":"Use What You Have: Video Retrieval Using Representations From Collaborative Experts","date":"2019-07-31","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":5,"samples_ran":0,"samples_unverified":5,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/multi-agent-reinforcement-learning-based","title":"Multi-Agent Reinforcement Learning Based Frame Sampling for Effective Untrimmed Video Recognition","date":"2019-07-31","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/bmn-boundary-matching-network-for-temporal","title":"BMN: Boundary-Matching Network for Temporal Action Proposal Generation","date":"2019-07-23","rows_on_this_dataset":2,"code_links":15,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":11,"samples_ran":9,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/completeness-modeling-and-context-separation","title":"Completeness Modeling and Context Separation for Weakly Supervised Temporal Action Localization","date":"2019-06-01","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/marginalized-average-attentional-network-for-1","title":"Marginalized Average Attentional Network for Weakly-Supervised Learning","date":"2019-05-21","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/multi-granularity-generator-for-temporal","title":"Multi-granularity Generator for Temporal Action Proposal","date":"2018-11-28","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/fine-grained-video-categorization-with","title":"Fine-grained Video Categorization with Redundancy Reduction Attention","date":"2018-10-26","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/autoloc-weakly-supervised-temporal-action-1","title":"AutoLoc: Weakly-supervised Temporal Action Localization in Untrimmed Videos","date":"2018-09-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/w-talc-weakly-supervised-temporal-activity","title":"W-TALC: Weakly-supervised Temporal Activity Localization and Classification","date":"2018-07-27","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":7,"samples_ran":0,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/ctap-complementary-temporal-action-proposal","title":"CTAP: Complementary Temporal Action Proposal Generation","date":"2018-07-12","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/bsn-boundary-sensitive-network-for-temporal","title":"BSN: Boundary Sensitive Network for Temporal Action Proposal Generation","date":"2018-06-08","rows_on_this_dataset":2,"code_links":17,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/towards-universal-representation-for-unseen","title":"Towards Universal Representation for Unseen Action Recognition","date":"2018-03-22","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/weakly-supervised-action-localization-by","title":"Weakly Supervised Action Localization by Sparse Temporal Pooling Network","date":"2017-12-14","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/learning-spatio-temporal-representation-with","title":"Learning Spatio-Temporal Representation with Pseudo-3D Residual Networks","date":"2017-11-28","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/temporal-convolution-based-action-proposal","title":"Temporal Convolution Based Action Proposal: Submission to ActivityNet 2017","date":"2017-07-21","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/untrimmednets-for-weakly-supervised-action","title":"UntrimmedNets for Weakly Supervised Action Recognition and Detection","date":"2017-03-09","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/a-pursuit-of-temporal-accuracy-in-general","title":"A Pursuit of Temporal Accuracy in General Activity Detection","date":"2017-03-08","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/youtube-8m-a-large-scale-video-classification","title":"YouTube-8M: A Large-Scale Video Classification Benchmark","date":"2016-09-27","rows_on_this_dataset":1,"code_links":7,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":9,"samples_ran":1,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/do-less-and-achieve-more-training-cnns-for","title":"Do Less and Achieve More: Training CNNs for Action Recognition Utilizing Action Images from the Web","date":"2015-12-22","rows_on_this_dataset":2,"code_links":0,"syntology":null}],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":37,"samples_harvested":302,"samples_ran":160,"samples_unverified":142,"pointer_only_for_licence":31,"papers_with_no_sample_that_ran":6,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}