{"url":"/dataset/msvd","name":"MSVD","full_name":"Microsoft Research Video Description Corpus","description_markdown":"The **Microsoft Research Video Description Corpus** (**MSVD**) dataset consists of about 120K sentences collected during the summer of 2010. Workers on Mechanical Turk were paid to watch a short video snippet and then summarize the action in a single sentence. The result is a set of roughly parallel descriptions of more than 2,000 video snippets. Because the workers were urged to complete the task in the language of their choice, both paraphrase and bilingual alternations are captured in the data.\r\n\r\nSource: [https://www.microsoft.com/en-us/download/details.aspx?id=52422&from=https%3A%2F%2Fresearch.microsoft.com%2Fen-us%2Fdownloads%2F38cf15fd-b8df-477e-a4e4-a4680caa75af%2F](https://www.microsoft.com/en-us/download/details.aspx?id=52422&from=https%3A%2F%2Fresearch.microsoft.com%2Fen-us%2Fdownloads%2F38cf15fd-b8df-477e-a4e4-a4680caa75af%2F)\r\nImage Source: [https://arxiv.org/pdf/1609.06782.pdf](https://arxiv.org/pdf/1609.06782.pdf)","description_withheld":null,"homepage":"https://www.cs.utexas.edu/users/ml/clamp/videoDescription/","introduced_date":"2011-01-01","introduced_date_note":null,"introduced_by":{"paper":null,"title":"Collecting Highly Parallel Data for Paraphrase Evaluation","first_author":null,"url":"https://www.aclweb.org/anthology/P11-1020/"},"license":{"name":"Unknown","url":null},"modalities":[{"name":"Videos","url":"/datasets/modality/videos"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Video Retrieval","url":"/task/video-retrieval","datasets_with_task":"/datasets/task/video-retrieval"},{"name":"Video Captioning","url":"/task/video-captioning","datasets_with_task":"/datasets/task/video-captioning"},{"name":"Zero-Shot Video Retrieval","url":"/task/zero-shot-video-retrieval","datasets_with_task":"/datasets/task/zero-shot-video-retrieval"}],"languages":[{"name":"Chinese","url":"/datasets/language/chinese"}],"variants":["MSVD"],"data_loaders":[],"num_papers_in_archive":327,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/video-retrieval-on-msvd","task":"Video Retrieval","dataset_variant":"MSVD","rows":24,"metrics":["text-to-video R@1","text-to-video R@5","text-to-video R@10","text-to-video Median Rank","text-to-video Mean Rank","text-to-video R@50","video-to-text R@1","video-to-text R@5","video-to-text R@10","video-to-text Median Rank","video-to-text Mean Rank"],"first_row_in_archive_order":{"model":"InternVideo2-6B","paper":"/paper/internvideo2-scaling-video-foundation-models","metrics":{"text-to-video R@1":"61.4","video-to-text R@1":"85.2"},"code_links":[{"title":"opengvlab/internvideo","url":"https://github.com/opengvlab/internvideo"},{"title":"opengvlab/internvideo2","url":"https://github.com/opengvlab/internvideo2"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/video-captioning-on-msvd-1","task":"Video Captioning","dataset_variant":"MSVD","rows":16,"metrics":["CIDEr","BLEU-4","METEOR","ROUGE-L","GS"],"first_row_in_archive_order":{"model":"MaMMUT","paper":"/paper/mammut-a-simple-architecture-for-joint","metrics":{"CIDEr":"195.6"},"code_links":[{"title":"lucidrains/mammut-pytorch","url":"https://github.com/lucidrains/mammut-pytorch"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/zero-shot-video-retrieval-on-msvd","task":"Zero-Shot Video Retrieval","dataset_variant":"MSVD","rows":14,"metrics":["text-to-video R@1","text-to-video R@5","text-to-video R@10","text-to-video Median Rank","text-to-video Mean Rank","video-to-text R@1","video-to-text R@5","video-to-text R@10","video-to-text Median Rank"],"first_row_in_archive_order":{"model":"InternVideo2-6B","paper":"/paper/internvideo2-scaling-video-foundation-models","metrics":{"text-to-video R@1":"59.3","text-to-video R@10":"89.6","text-to-video R@5":"84.4","video-to-text R@1":"83.1","video-to-text R@10":"97.0","video-to-text R@5":"94.2"},"code_links":[{"title":"opengvlab/internvideo","url":"https://github.com/opengvlab/internvideo"},{"title":"opengvlab/internvideo2","url":"https://github.com/opengvlab/internvideo2"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/internvideo2-scaling-video-foundation-models","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","date":"2024-03-22","rows_on_this_dataset":3,"code_links":2,"syntology":null},{"paper":"/paper/vid-tldr-training-free-token-merging-for","title":"vid-TLDR: Training Free Token merging for Light-weight Video Transformer","date":"2024-03-20","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/rtq-rethinking-video-language-understanding","title":"RTQ: Rethinking Video-language Understanding Based on Image-text Model","date":"2023-12-01","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/side4video-spatial-temporal-side-network-for","title":"Side4Video: Spatial-Temporal Side Network for Memory-Efficient Image-to-Video Transfer Learning","date":"2023-11-27","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/howtocaption-prompting-llms-to-transform","title":"HowToCaption: Prompting LLMs to Transform Video Annotations at Scale","date":"2023-10-07","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":4,"samples_unverified":1,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/icocap-improving-video-captioning-by","title":"IcoCap: Improving Video Captioning by Compounding Images","date":"2023-10-05","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/languagebind-extending-video-language","title":"LanguageBind: Extending Video-Language Pretraining to N-modality by Language-based Semantic Alignment","date":"2023-10-03","rows_on_this_dataset":2,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":14,"samples_ran":7,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/prototype-based-aleatoric-uncertainty-1","title":"Prototype-based Aleatoric Uncertainty Quantification for Cross-modal Retrieval","date":"2023-09-29","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":19,"samples_ran":12,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/accurate-and-fast-compressed-video-captioning","title":"Accurate and Fast Compressed Video Captioning","date":"2023-09-22","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":14,"samples_ran":10,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/dual-modal-attention-enhanced-text-video","title":"Dual-Modal Attention-Enhanced Text-Video Retrieval with Triplet Partial Margin Contrastive Learning","date":"2023-09-20","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/cosa-concatenated-sample-pretrained-vision","title":"COSA: Concatenated Sample Pretrained Vision-Language Foundation Model","date":"2023-06-15","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/vlab-enhancing-video-language-pre-training-by","title":"VLAB: Enhancing Video Language Pre-training by Feature Adapting and Blending","date":"2023-05-22","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/valor-vision-audio-language-omni-perception","title":"VALOR: Vision-Audio-Language Omni-Perception Pretraining Model and Dataset","date":"2023-04-17","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/mammut-a-simple-architecture-for-joint","title":"MaMMUT: A Simple Architecture for Joint Learning for MultiModal Tasks","date":"2023-03-29","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/unmasked-teacher-towards-training-efficient","title":"Unmasked Teacher: Towards Training-Efficient Video Foundation Models","date":"2023-03-28","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":3,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/sem-pos-grammatically-and-semantically","title":"SEM-POS: Grammatically and Semantically Correct Video Captioning","date":"2023-03-26","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/diffusionret-generative-text-video-retrieval","title":"DiffusionRet: Generative Text-Video Retrieval with Diffusion Model","date":"2023-03-17","rows_on_this_dataset":2,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":5,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vid2seq-large-scale-pretraining-of-a-visual","title":"Vid2Seq: Large-Scale Pretraining of a Visual Language Model for Dense Video Captioning","date":"2023-02-27","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/mplug-2-a-modularized-multi-modal-foundation","title":"mPLUG-2: A Modularized Multi-modal Foundation Model Across Text, Image and Video","date":"2023-02-01","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":19,"samples_ran":9,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/cap4video-what-can-auxiliary-captions-do-for","title":"Cap4Video: What Can Auxiliary Captions Do for Text-Video Retrieval?","date":"2022-12-31","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":25,"samples_ran":14,"samples_unverified":11,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/hitea-hierarchical-temporal-aware-video","title":"HiTeA: Hierarchical Temporal-Aware Video-Language Pre-training","date":"2022-12-30","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/internvideo-general-video-foundation-models","title":"InternVideo: General Video Foundation Models via Generative and Discriminative Learning","date":"2022-12-06","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/an-empirical-study-of-end-to-end-video","title":"An Empirical Study of End-to-End Video-Language Transformers with Masked Visual Modeling","date":"2022-09-04","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/diverse-video-captioning-by-adaptive-spatio","title":"Diverse Video Captioning by Adaptive Spatio-temporal Attention","date":"2022-08-19","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/x-clip-end-to-end-multi-grained-contrastive","title":"X-CLIP: End-to-End Multi-grained Contrastive Learning for Video-Text Retrieval","date":"2022-07-15","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/lat-latent-translation-with-cycle-consistency","title":"LaT: Latent Translation with Cycle-Consistency for Video-Text Retrieval","date":"2022-07-11","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/centerclip-token-clustering-for-efficient","title":"CenterCLIP: Token Clustering for Efficient Text-Video Retrieval","date":"2022-05-02","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/miles-visual-bert-pre-training-with-injected","title":"MILES: Visual BERT Pre-training with Injected Language Semantics for Video-text Retrieval","date":"2022-04-26","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/hunyuan-tvr-for-text-video-retrivial","title":"Tencent Text-Video Retrieval: Hierarchical Cross-Modal Interactions with Multi-Level Representations","date":"2022-04-07","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/x-pool-cross-modal-language-video-attention","title":"X-Pool: Cross-Modal Language-Video Attention for Text-Video Retrieval","date":"2022-03-28","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mdmmt-2-multidomain-multimodal-transformer","title":"MDMMT-2: Multidomain Multimodal Transformer for Video Retrieval, One More Step Towards Generalization","date":"2022-03-14","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/bridgeformer-bridging-video-text-retrieval","title":"Bridging Video-text Retrieval with Multiple Choice Questions","date":"2022-01-13","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":24,"samples_ran":13,"samples_unverified":11,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/cross-modal-retrieval-with-querybank","title":"Cross Modal Retrieval with Querybank Normalisation","date":"2021-12-23","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/lightweight-attentional-feature-fusion-for","title":"Lightweight Attentional Feature Fusion: A New Baseline for Text-to-Video Retrieval","date":"2021-12-03","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/improving-video-text-retrieval-by-multi","title":"Improving Video-Text Retrieval by Multi-Stream Corpus Alignment and Dual Softmax Loss","date":"2021-09-09","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/clip4clip-an-empirical-study-of-clip-for-end","title":"CLIP4Clip: An Empirical Study of CLIP for End to End Video Clip Retrieval","date":"2021-04-18","rows_on_this_dataset":2,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/frozen-in-time-a-joint-video-and-image","title":"Frozen in Time: A Joint Video and Image Encoder for End-to-End Retrieval","date":"2021-04-01","rows_on_this_dataset":1,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":3,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/a-straightforward-framework-for-video","title":"A Straightforward Framework For Video Retrieval Using CLIP","date":"2021-02-24","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/noise-estimation-using-density-estimation-for","title":"Noise Estimation Using Density Estimation for Self-Supervised Multimodal Learning","date":"2020-03-06","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/use-what-you-have-video-retrieval-using","title":"Use What You Have: Video Retrieval Using Representations From Collaborative Experts","date":"2019-07-31","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":0,"samples_unverified":5,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":19,"samples_harvested":172,"samples_ran":96,"samples_unverified":76,"pointer_only_for_licence":19,"papers_with_no_sample_that_ran":2,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}