{"url":"/dataset/activitynet-captions","name":"ActivityNet Captions","full_name":null,"description_markdown":"The **ActivityNet Captions** dataset is built on ActivityNet v1.3 which includes 20k YouTube untrimmed videos with 100k caption annotations. The videos are 120 seconds long on average. Most of the videos contain over 3 annotated events with corresponding start/end time and human-written sentences, which contain 13.5 words on average. The number of videos in train/validation/test split is 10024/4926/5044, respectively.\r\n\r\nSource: [Bidirectional Attentive Fusion with Context Gating for Dense Video Captioning](https://arxiv.org/abs/1804.00100)\r\nImage Source: [https://cs.stanford.edu/people/ranjaykrishna/densevid/](https://cs.stanford.edu/people/ranjaykrishna/densevid/)","description_withheld":null,"homepage":"https://cs.stanford.edu/people/ranjaykrishna/densevid/","introduced_date":"2017-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/dense-captioning-events-in-videos","title":"Dense-Captioning Events in Videos","first_author":"Ranjay Krishna","url":null},"license":{"name":"Unknown","url":null},"modalities":[{"name":"Videos","url":"/datasets/modality/videos"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Video Captioning","url":"/task/video-captioning","datasets_with_task":"/datasets/task/video-captioning"},{"name":"Dense Video Captioning","url":"/task/dense-video-captioning","datasets_with_task":"/datasets/task/dense-video-captioning"},{"name":"Natural Language Moment Retrieval","url":"/task/natural-language-moment-retrieval","datasets_with_task":"/datasets/task/natural-language-moment-retrieval"},{"name":"Temporal Action Proposal Generation","url":"/task/temporal-action-proposal-generation","datasets_with_task":"/datasets/task/temporal-action-proposal-generation"},{"name":"Partially Relevant Video Retrieval","url":"/task/partially-relevant-video-retrieval","datasets_with_task":"/datasets/task/partially-relevant-video-retrieval"},{"name":"Live Video Captioning","url":"/task/live-video-captioning","datasets_with_task":"/datasets/task/live-video-captioning"}],"languages":[],"variants":["ActivityNet Captions"],"data_loaders":[],"num_papers_in_archive":255,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/dense-video-captioning-on-activitynet","task":"Dense Video Captioning","dataset_variant":"ActivityNet Captions","rows":12,"metrics":["METEOR","BLEU-3","BLEU-4","CIDEr","SODA","DIV-1","DIV-2","RE-4","BLEU4","F1","Precision","Recall"],"first_row_in_archive_order":{"model":"Vid2Seq","paper":"/paper/vid2seq-large-scale-pretraining-of-a-visual","metrics":{"CIDEr":"28","METEOR":"17"},"code_links":[{"title":"google-research/scenic","url":"https://github.com/google-research/scenic/tree/main/scenic/projects/vid2seq"},{"title":"antoyang/VidChapters","url":"https://github.com/antoyang/VidChapters"},{"title":"KastanDay/video-pretrained-transformer","url":"https://github.com/KastanDay/video-pretrained-transformer"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/natural-language-moment-retrieval-on","task":"Natural Language Moment Retrieval","dataset_variant":"ActivityNet Captions","rows":8,"metrics":["R@1,IoU=0.5","R@1,IoU=0.7","R@5,IoU=0.5","R@5,IoU=0.7"],"first_row_in_archive_order":{"model":"GVL (paragraph-level)","paper":"/paper/learning-grounded-vision-language","metrics":{"R@1,IoU=0.5":"60.67","R@1,IoU=0.7":"38.55"},"code_links":[{"title":"zjr2000/gvl","url":"https://github.com/zjr2000/gvl"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/video-captioning-on-activitynet-captions","task":"Video Captioning","dataset_variant":"ActivityNet Captions","rows":5,"metrics":["BLEU4","BLEU-3","CIDEr","ROUGE-L","METEOR"],"first_row_in_archive_order":{"model":"VideoCoCa","paper":"/paper/video-text-modeling-with-zero-shot-transfer","metrics":{"BLEU4":"14.7","CIDEr":"39.3","ROUGE-L":"35.0"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/live-video-captioning-on-activitynet-captions","task":"Live Video Captioning","dataset_variant":"ActivityNet Captions","rows":1,"metrics":["Live Score"],"first_row_in_archive_order":{"model":"LVC","paper":"/paper/live-video-captioning","metrics":{"Live Score":"20.81"},"code_links":[{"title":"gramuah/lvc","url":"https://github.com/gramuah/lvc"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/partially-relevant-video-retrieval-on","task":"Partially Relevant Video Retrieval","dataset_variant":"ActivityNet Captions","rows":1,"metrics":["Recall@Sum"],"first_row_in_archive_order":{"model":"ms-sl","paper":"/paper/partially-relevant-video-retrieval","metrics":{"Recall@Sum":"140.1"},"code_links":[{"title":"HuiGuanLab/ms-sl","url":"https://github.com/HuiGuanLab/ms-sl"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/temporal-action-proposal-generation-on-1","task":"Temporal Action Proposal Generation","dataset_variant":"ActivityNet Captions","rows":1,"metrics":["Average F1","Average Precision","Average Recall"],"first_row_in_archive_order":{"model":"BMT","paper":"/paper/a-better-use-of-audio-visual-cues-dense-video","metrics":{"Average F1":"60.27","Average Precision":"48.23","Average Recall":"80.31"},"code_links":[{"title":"v-iashin/video_features","url":"https://github.com/v-iashin/video_features"},{"title":"v-iashin/BMT","url":"https://github.com/v-iashin/BMT"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/llava-mr-large-language-and-vision-assistant","title":"LLaVA-MR: Large Language-and-Vision Assistant for Video Moment Retrieval","date":"2024-11-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/live-video-captioning","title":"Live Video Captioning","date":"2024-06-20","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/do-you-remember-dense-video-captioning-with","title":"Do You Remember? Dense Video Captioning with Cross-Modal Memory Retrieval","date":"2024-04-11","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/unimd-towards-unifying-moment-retrieval-and","title":"UniMD: Towards Unifying Moment Retrieval and Temporal Action Detection","date":"2024-04-07","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/vtimellm-empower-llm-to-grasp-video-moments","title":"VTimeLLM: Empower LLM to Grasp Video Moments","date":"2023-11-30","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":5,"samples_unverified":6,"pointer_only_for_licence":11,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/unloc-a-unified-framework-for-video","title":"UnLoc: A Unified Framework for Video Localization Tasks","date":"2023-08-21","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/learning-grounded-vision-language","title":"Learning Grounded Vision-Language Representation for Versatile Understanding in Untrimmed Videos","date":"2023-03-11","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":1,"samples_unverified":11,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vid2seq-large-scale-pretraining-of-a-visual","title":"Vid2Seq: Large-Scale Pretraining of a Visual Language Model for Dense Video Captioning","date":"2023-02-27","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/video-text-modeling-with-zero-shot-transfer","title":"VideoCoCa: Video-Text Modeling with Zero-Shot Transfer from Contrastive Captioners","date":"2022-12-09","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/vltint-visual-linguistic-transformer-in","title":"VLTinT: Visual-Linguistic Transformer-in-Transformer for Coherent Video Paragraph Captioning","date":"2022-11-28","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/partially-relevant-video-retrieval","title":"Partially Relevant Video Retrieval","date":"2022-08-26","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/vlcap-vision-language-with-contrastive","title":"VLCap: Vision-Language with Contrastive Learning for Coherent Video Paragraph Captioning","date":"2022-06-26","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/end-to-end-dense-video-captioning-with","title":"End-to-End Dense Video Captioning with Parallel Decoding","date":"2021-08-17","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":3,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/global-object-proposals-for-improving-multi","title":"Global Object Proposals for Improving Multi-Sentence Video Descriptions","date":"2021-07-18","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/tsp-temporally-sensitive-pretraining-of-video","title":"TSP: Temporally-Sensitive Pretraining of Video Encoders for Localization Tasks","date":"2020-11-23","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/vlg-net-video-language-graph-matching-network","title":"VLG-Net: Video-Language Graph Matching Network for Video Grounding","date":"2020-11-19","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/iperceive-applying-common-sense-reasoning-to-1","title":"iPerceive: Applying Common-Sense Reasoning to Multi-Modal Dense Video Captioning and Video Question Answering","date":"2020-11-16","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/coot-cooperative-hierarchical-transformer-for","title":"COOT: Cooperative Hierarchical Transformer for Video-Text Representation Learning","date":"2020-11-01","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":0,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/dense-captioning-events-in-videos-sysu","title":"Dense-Captioning Events in Videos: SYSU Submission to ActivityNet Challenge 2020","date":"2020-06-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/team-ruc-aim3-technical-report-at-activitynet","title":"Team RUC_AIM3 Technical Report at Activitynet 2020 Task 2: Exploring Sequential Events Detection for Dense Video Captioning","date":"2020-06-14","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/a-better-use-of-audio-visual-cues-dense-video","title":"A Better Use of Audio-Visual Cues: Dense Video Captioning with Bi-modal Transformer","date":"2020-05-17","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":0,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mart-memory-augmented-recurrent-transformer","title":"MART: Memory-Augmented Recurrent Transformer for Coherent Video Paragraph Captioning","date":"2020-05-11","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/dense-regression-network-for-video-grounding","title":"Dense Regression Network for Video Grounding","date":"2020-04-07","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/multi-modal-dense-video-captioning","title":"Multi-modal Dense Video Captioning","date":"2020-03-17","rows_on_this_dataset":1,"code_links":4,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":8,"samples_harvested":49,"samples_ran":13,"samples_unverified":36,"pointer_only_for_licence":11,"papers_with_no_sample_that_ran":2,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}