{"url":"/dataset/queryd","name":"QuerYD","full_name":null,"description_markdown":"A large-scale dataset for retrieval and event localisation in video. A unique feature of the dataset is the availability of two audio tracks for each video: the original audio, and a high-quality spoken description of the visual content.\r\n\r\nSource: [QuerYD: A video dataset with high-quality textual and audio narrations](/paper/queryd-a-video-dataset-with-high-quality)","description_withheld":null,"homepage":"https://www.robots.ox.ac.uk/~vgg/data/queryd/","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/queryd-a-video-dataset-with-high-quality","title":"QuerYD: A video dataset with high-quality text and audio narrations","first_author":"Andreea-Maria Oncescu","url":null},"license":null,"modalities":[{"name":"Videos","url":"/datasets/modality/videos"},{"name":"Texts","url":"/datasets/modality/texts"},{"name":"Audio","url":"/datasets/modality/audio"}],"tasks":[{"name":"Video Retrieval","url":"/task/video-retrieval","datasets_with_task":"/datasets/task/video-retrieval"},{"name":"Video Understanding","url":"/task/video-understanding","datasets_with_task":"/datasets/task/video-understanding"}],"languages":[],"variants":["QuerYD"],"data_loaders":[],"num_papers_in_archive":15,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/video-retrieval-on-queryd","task":"Video Retrieval","dataset_variant":"QuerYD","rows":5,"metrics":["text-to-video R@1","text-to-video R@5","text-to-video R@10"],"first_row_in_archive_order":{"model":"TESTA (ViT-B/16)","paper":"/paper/testa-temporal-spatial-token-aggregation-for","metrics":{"text-to-video R@1":"83.4","text-to-video R@10":"95.3","text-to-video R@5":"93.8"},"code_links":[{"title":"renshuhuai-andy/testa","url":"https://github.com/renshuhuai-andy/testa"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/testa-temporal-spatial-token-aggregation-for","title":"TESTA: Temporal-Spatial Token Aggregation for Long-form Video-Language Understanding","date":"2023-10-29","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":11,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vindlu-a-recipe-for-effective-video-and","title":"VindLU: A Recipe for Effective Video-and-Language Pretraining","date":"2022-12-09","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/long-form-video-language-pre-training-with","title":"Long-Form Video-Language Pre-Training with Multimodal Temporal Contrastive Learning","date":"2022-10-12","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/cross-modal-retrieval-with-querybank","title":"Cross Modal Retrieval with Querybank Normalisation","date":"2021-12-23","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/frozen-in-time-a-joint-video-and-image","title":"Frozen in Time: A Joint Video and Image Encoder for End-to-End Retrieval","date":"2021-04-01","rows_on_this_dataset":1,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":3,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":4,"samples_harvested":30,"samples_ran":17,"samples_unverified":13,"pointer_only_for_licence":1,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}