{"url":"/dataset/vatex","name":"VATEX","full_name":"Video And TEXt","description_markdown":"**VATEX** is multilingual, large, linguistically complex, and diverse dataset in terms of both video and natural language descriptions. It has two tasks for video-and-language research: (1) Multilingual Video Captioning, aimed at describing a video in various languages with a compact unified captioning model, and (2) Video-guided Machine Translation, to translate a source language description into the target language using the video information as additional spatiotemporal context.\r\n\r\nSource: [https://arxiv.org/pdf/1904.03493.pdf](https://arxiv.org/pdf/1904.03493.pdf)\r\nImage Source: [https://arxiv.org/pdf/1904.03493.pdf](https://arxiv.org/pdf/1904.03493.pdf)","description_withheld":null,"homepage":"https://eric-xw.github.io/vatex-website/index.html","introduced_date":"2019-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/vatex-a-large-scale-high-quality-multilingual","title":"VATEX: A Large-Scale, High-Quality Multilingual Dataset for Video-and-Language Research","first_author":"Xin Wang","url":null},"license":{"name":"CC BY 4.0","url":"https://creativecommons.org/licenses/by/4.0/"},"modalities":[{"name":"Videos","url":"/datasets/modality/videos"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Video Retrieval","url":"/task/video-retrieval","datasets_with_task":"/datasets/task/video-retrieval"},{"name":"Video Captioning","url":"/task/video-captioning","datasets_with_task":"/datasets/task/video-captioning"},{"name":"Zero-Shot Video Retrieval","url":"/task/zero-shot-video-retrieval","datasets_with_task":"/datasets/task/zero-shot-video-retrieval"}],"languages":[{"name":"English","url":"/datasets/language/english"},{"name":"Chinese","url":"/datasets/language/chinese"}],"variants":["VATEX"],"data_loaders":[],"num_papers_in_archive":118,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/video-retrieval-on-vatex","task":"Video Retrieval","dataset_variant":"VATEX","rows":13,"metrics":["text-to-video R@1","text-to-video R@5","text-to-video R@10","text-to-video R@50","text-to-video MedianR","text-to-video MeanR","video-to-text R@1","video-to-text R@10"],"first_row_in_archive_order":{"model":"GRAM","paper":"/paper/gramian-multimodal-representation-learning","metrics":{"text-to-video R@1":"87.7","text-to-video R@10":"100","video-to-text R@1":"84.6","video-to-text R@10":"100"},"code_links":[{"title":"ispamm/GRAM","url":"https://github.com/ispamm/GRAM"},{"title":"luigisigillo/gwit","url":"https://github.com/luigisigillo/gwit"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/video-captioning-on-vatex-1","task":"Video Captioning","dataset_variant":"VATEX","rows":10,"metrics":["BLEU-4","CIDEr","METEOR","ROUGE-L"],"first_row_in_archive_order":{"model":"VALOR","paper":"/paper/valor-vision-audio-language-omni-perception","metrics":{"BLEU-4":"45.6","CIDEr":"95.8","METEOR":"29.4","ROUGE-L":"57.4"},"code_links":[{"title":"TXH-mercury/VALOR","url":"https://github.com/TXH-mercury/VALOR"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/zero-shot-video-retrieval-on-vatex","task":"Zero-Shot Video Retrieval","dataset_variant":"VATEX","rows":5,"metrics":["text-to-video R@1","text-to-video R@5","text-to-video R@10","video-to-text R@1","video-to-text R@5","video-to-text R@10"],"first_row_in_archive_order":{"model":"GRAM","paper":"/paper/gramian-multimodal-representation-learning","metrics":{"text-to-video R@1":"83.9","text-to-video R@10":"99.5","video-to-text R@1":"82.7","video-to-text R@10":"99"},"code_links":[{"title":"ispamm/GRAM","url":"https://github.com/ispamm/GRAM"},{"title":"luigisigillo/gwit","url":"https://github.com/luigisigillo/gwit"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/gramian-multimodal-representation-learning","title":"Gramian Multimodal Representation Learning and Alignment","date":"2024-12-16","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":2,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/internvideo2-scaling-video-foundation-models","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","date":"2024-03-22","rows_on_this_dataset":3,"code_links":2,"syntology":null},{"paper":"/paper/holistic-features-are-almost-sufficient-for","title":"Holistic Features are almost Sufficient for Text-to-Video Retrieval","date":"2024-01-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/side4video-spatial-temporal-side-network-for","title":"Side4Video: Spatial-Temporal Side Network for Memory-Efficient Image-to-Video Transfer Learning","date":"2023-11-27","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/icocap-improving-video-captioning-by","title":"IcoCap: Improving Video Captioning by Compounding Images","date":"2023-10-05","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/accurate-and-fast-compressed-video-captioning","title":"Accurate and Fast Compressed Video Captioning","date":"2023-09-22","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":14,"samples_ran":10,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/cosa-concatenated-sample-pretrained-vision","title":"COSA: Concatenated Sample Pretrained Vision-Language Foundation Model","date":"2023-06-15","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/vast-a-vision-audio-subtitle-text-omni-1","title":"VAST: A Vision-Audio-Subtitle-Text Omni-Modality Foundation Model and Dataset","date":"2023-05-29","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":42,"samples_ran":15,"samples_unverified":27,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/valor-vision-audio-language-omni-perception","title":"VALOR: Vision-Audio-Language Omni-Perception Pretraining Model and Dataset","date":"2023-04-17","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/unmasked-teacher-towards-training-efficient","title":"Unmasked Teacher: Towards Training-Efficient Video Foundation Models","date":"2023-03-28","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":3,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/cap4video-what-can-auxiliary-captions-do-for","title":"Cap4Video: What Can Auxiliary Captions Do for Text-Video Retrieval?","date":"2022-12-31","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":25,"samples_ran":14,"samples_unverified":11,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/video-text-modeling-with-zero-shot-transfer","title":"VideoCoCa: Video-Text Modeling with Zero-Shot Transfer from Contrastive Captioners","date":"2022-12-09","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/internvideo-general-video-foundation-models","title":"InternVideo: General Video Foundation Models via Generative and Discriminative Learning","date":"2022-12-06","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/diverse-video-captioning-by-adaptive-spatio","title":"Diverse Video Captioning by Adaptive Spatio-temporal Attention","date":"2022-08-19","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/ts2-net-token-shift-and-selection-transformer","title":"TS2-Net: Token Shift and Selection Transformer for Text-Video Retrieval","date":"2022-07-16","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":4,"samples_unverified":0,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/cross-modal-retrieval-with-querybank","title":"Cross Modal Retrieval with Querybank Normalisation","date":"2021-12-23","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/lightweight-attentional-feature-fusion-for","title":"Lightweight Attentional Feature Fusion: A New Baseline for Text-to-Video Retrieval","date":"2021-12-03","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/clip2video-mastering-video-text-retrieval-via","title":"CLIP2Video: Mastering Video-Text Retrieval via Image CLIP","date":"2021-06-21","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/nits-vc-system-for-vatex-video-captioning","title":"NITS-VC System for VATEX Video Captioning Challenge 2020","date":"2020-06-07","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/object-relational-graph-with-teacher","title":"Object Relational Graph with Teacher-Recommended Learning for Video Captioning","date":"2020-02-26","rows_on_this_dataset":1,"code_links":0,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":9,"samples_harvested":112,"samples_ran":53,"samples_unverified":59,"pointer_only_for_licence":4,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}