{"url":"/dataset/howto100m","name":"HowTo100M","full_name":"HowTo100M","description_markdown":"HowTo100M is a large-scale dataset of narrated videos with an emphasis on instructional videos where content creators teach complex tasks with an explicit intention of explaining the visual content on screen. HowTo100M features a total of:\r\n\r\n- 136M video clips with captions sourced from 1.2M Youtube videos (15 years of video)\r\n- 23k activities from domains such as cooking, hand crafting, personal care, gardening or fitness\r\n\r\nEach video is associated with a narration available as subtitles automatically downloaded from Youtube.\r\n\r\nSource: [HowTo100M](https://www.di.ens.fr/willow/research/howto100m/)","description_withheld":null,"homepage":"https://www.di.ens.fr/willow/research/howto100m/","introduced_date":"2019-06-07","introduced_date_note":null,"introduced_by":{"paper":"/paper/howto100m-learning-a-text-video-embedding-by","title":"HowTo100M: Learning a Text-Video Embedding by Watching Hundred Million Narrated Video Clips","first_author":"Antoine Miech","url":null},"license":{"name":"Custom","url":"https://www.di.ens.fr/willow/research/howto100m/"},"modalities":[{"name":"Videos","url":"/datasets/modality/videos"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Action Recognition","url":"/task/action-recognition-in-videos","datasets_with_task":"/datasets/task/action-recognition-in-videos"},{"name":"Video Question Answering","url":"/task/video-question-answering","datasets_with_task":"/datasets/task/video-question-answering"},{"name":"Video Retrieval","url":"/task/video-retrieval","datasets_with_task":"/datasets/task/video-retrieval"},{"name":"Video Captioning","url":"/task/video-captioning","datasets_with_task":"/datasets/task/video-captioning"}],"languages":[],"variants":["Howto100M-QA","HowTo100M"],"data_loaders":[],"num_papers_in_archive":286,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/video-question-answering-on-howto100m-qa","task":"Video Question Answering","dataset_variant":"Howto100M-QA","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"TimeSformer","paper":"/paper/is-space-time-attention-all-you-need-for","metrics":{"Accuracy":"62.1"},"code_links":[{"title":"open-mmlab/mmaction2","url":"https://github.com/open-mmlab/mmaction2"},{"title":"towhee-io/towhee","url":"https://github.com/towhee-io/towhee"},{"title":"facebookresearch/TimeSformer","url":"https://github.com/facebookresearch/TimeSformer"},{"title":"PaddlePaddle/PaddleVideo","url":"https://github.com/PaddlePaddle/PaddleVideo/blob/develop/docs/zh-CN/model_zoo/recognition/timesformer.md"},{"title":"The-AI-Summer/self-attention-cv","url":"https://github.com/The-AI-Summer/self-attention-cv"},{"title":"lucidrains/TimeSformer-pytorch","url":"https://github.com/lucidrains/TimeSformer-pytorch"},{"title":"mx-mark/videotransformer-pytorch","url":"https://github.com/mx-mark/videotransformer-pytorch"},{"title":"jerrywn121/TianChi_AIEarth","url":"https://github.com/jerrywn121/TianChi_AIEarth"},{"title":"davide-coccomini/TimeSformer-Video-Classification","url":"https://github.com/davide-coccomini/TimeSformer-Video-Classification"},{"title":"yiyixuxu/TimeSformer-rolled-attention","url":"https://github.com/yiyixuxu/TimeSformer-rolled-attention"},{"title":"m-bain/video-transformers","url":"https://github.com/m-bain/video-transformers"},{"title":"md-mohaiminul/objectstatechange","url":"https://github.com/md-mohaiminul/objectstatechange"},{"title":"MindCode-4/code-5","url":"https://github.com/MindCode-4/code-5/tree/main/timesformer"},{"title":"pwc-1/Paper-9","url":"https://github.com/pwc-1/Paper-9/tree/main/1/timesformer"},{"title":"pwc-1/Paper-10","url":"https://github.com/pwc-1/Paper-10/tree/main/timesformer"},{"title":"halixness/generative_timesformer_pytorch","url":"https://github.com/halixness/generative_timesformer_pytorch"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/is-space-time-attention-all-you-need-for","title":"Is Space-Time Attention All You Need for Video Understanding?","date":"2021-02-09","rows_on_this_dataset":1,"code_links":16,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":43,"samples_ran":35,"samples_unverified":8,"pointer_only_for_licence":14,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":43,"samples_ran":35,"samples_unverified":8,"pointer_only_for_licence":14,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}