{"url":"/dataset/webvid-covr","name":"WebVid-CoVR","full_name":null,"description_markdown":"The WebVid-CoVR dataset is a collection of video-text-video triplets that can be used for the task of composed video retrieval (CoVR). CoVR is a task that involves searching for videos that match both a query image and a query text. The text typically specifies the desired modification to the query image. \r\n\r\nThe WebVid-CoVR dataset is automatically generated from web-scraped video-caption pairs, using a language model to generate the modification text. The dataset contains 1.6 million triplets, with diverse content and variations. The dataset also includes a manually annotated test set of 2.5K triplets, which can be used to evaluate CoVR models.","description_withheld":null,"homepage":"https://imagine.enpc.fr/~ventural/covr/","introduced_date":"2023-08-28","introduced_date_note":null,"introduced_by":{"paper":"/paper/covr-learning-composed-video-retrieval-from","title":"CoVR-2: Automatic Data Construction for Composed Video Retrieval","first_author":"Lucas Ventura","url":null},"license":null,"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Videos","url":"/datasets/modality/videos"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Composed Image Retrieval (CoIR)","url":"/task/composed-image-retrieval","datasets_with_task":"/datasets/task/composed-image-retrieval"},{"name":"Zero-Shot Composed Image Retrieval (ZS-CIR)","url":"/task/zero-shot-composed-image-retrieval-zs-cir","datasets_with_task":"/datasets/task/zero-shot-composed-image-retrieval-zs-cir"},{"name":"Composed Video Retrieval (CoVR)","url":"/task/composed-video-retrieval-covr","datasets_with_task":"/datasets/task/composed-video-retrieval-covr"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["WebVid-CoVR"],"data_loaders":[],"num_papers_in_archive":3,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/composed-video-retrieval-covr-on-covr","task":"Composed Video Retrieval (CoVR)","dataset_variant":"WebVid-CoVR","rows":1,"metrics":["R@1"],"first_row_in_archive_order":{"model":"BLIP-2","paper":"/paper/covr-learning-composed-video-retrieval-from","metrics":{"R@1":"59.82"},"code_links":[{"title":"lucas-ventura/CoVR","url":"https://github.com/lucas-ventura/CoVR"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/covr-learning-composed-video-retrieval-from","title":"CoVR-2: Automatic Data Construction for Composed Video Retrieval","date":"2023-08-28","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}