{"url":"/dataset/webvid","name":"WebVid","full_name":null,"description_markdown":"WebVid contains 10 million video clips with captions, sourced from the web. The videos are diverse and rich in their content.\r\n\r\nBoth the full 10M set and a 2.5M subset is available for download:\r\nhttps://github.com/m-bain/webvid-dataset","description_withheld":null,"homepage":"https://github.com/m-bain/webvid","introduced_date":"2021-04-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/frozen-in-time-a-joint-video-and-image","title":"Frozen in Time: A Joint Video and Image Encoder for End-to-End Retrieval","first_author":"Max Bain","url":null},"license":{"name":"Custom","url":"https://github.com/m-bain/webvid/blob/main/TERMS.md"},"modalities":[{"name":"Videos","url":"/datasets/modality/videos"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Video Retrieval","url":"/task/video-retrieval","datasets_with_task":"/datasets/task/video-retrieval"},{"name":"Video Captioning","url":"/task/video-captioning","datasets_with_task":"/datasets/task/video-captioning"},{"name":"Video Generation","url":"/task/video-generation","datasets_with_task":"/datasets/task/video-generation"},{"name":"Text-to-Video Generation","url":"/task/text-to-video-generation","datasets_with_task":"/datasets/task/text-to-video-generation"},{"name":"Video-Text Retrieval","url":"/task/video-text-retrieval","datasets_with_task":"/datasets/task/video-text-retrieval"}],"languages":[],"variants":["WebVid"],"data_loaders":[],"num_papers_in_archive":257,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/text-to-video-generation-on-webvid","task":"Text-to-Video Generation","dataset_variant":"WebVid","rows":1,"metrics":["FVD"],"first_row_in_archive_order":{"model":"VideoFactory","paper":"/paper/videofactory-swap-attention-in-spatiotemporal","metrics":{"FVD":"292.35"},"code_links":[{"title":"daooshee/hd-vg-130m","url":"https://github.com/daooshee/hd-vg-130m"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/videofactory-swap-attention-in-spatiotemporal","title":"Swap Attention in Spatiotemporal Diffusions for Text-to-Video Generation","date":"2023-05-18","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}