{"url":"/dataset/devan","name":"DeVAn","full_name":"Dense Video Annotation for Video-Language Models","description_markdown":"DeVAn is a multi-modal dataset containing 8.5K video clips carefully selected from previously published YouTube-based video datasets (YouTube-8M and YT-Temporal-1B) that integrate visual and auditory information. Over the span of 10 months, a team of 24 human annotators (college and graduate level students) created 5 short captions (1 sentence each) and 5 long summaries (3-10 sentences) for each video clip, resulting in a rich and comprehensive human-annotated dataset that serves as a robust ground truth for subsequent model training and evaluation.","description_withheld":null,"homepage":"https://www.tingkai-liu.org/DeVAn/","introduced_date":"2024-08-11","introduced_date_note":null,"introduced_by":{"paper":"/paper/video-csr-complex-video-digest-creation-for","title":"DeVAn: Dense Video Annotation for Video-Language Models","first_author":"Tingkai Liu","url":null},"license":null,"modalities":[{"name":"Videos","url":"/datasets/modality/videos"}],"tasks":[{"name":"Video Description","url":"/task/video-description","datasets_with_task":"/datasets/task/video-description"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["DeVAn"],"data_loaders":[],"num_papers_in_archive":1,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}