{"url":"/dataset/msrvtt-mc","name":"MSRVTT-MC","full_name":null,"description_markdown":"The MSRVTT-MC (Multiple Choice) dataset is a video question-answering dataset created based on the MSR-VTT dataset. It consists of 2,990 questions generated from 10,000 video clips with associated ground truth captions. For each question, there are five candidate captions, including the ground truth caption and four randomly sampled negative choices. The objective of the dataset is to choose the correct answer from the five candidate captions.","description_withheld":null,"homepage":"https://github.com/yj-yu/lsmdc","introduced_date":"2018-08-07","introduced_date_note":null,"introduced_by":{"paper":"/paper/a-joint-sequence-fusion-model-for-video","title":"A Joint Sequence Fusion Model for Video Question Answering and Retrieval","first_author":"Youngjae Yu","url":null},"license":null,"modalities":[],"tasks":[{"name":"Video Question Answering","url":"/task/video-question-answering","datasets_with_task":"/datasets/task/video-question-answering"}],"languages":[],"variants":["MSRVTT-MC","MSR-VTT-MC"],"data_loaders":[],"num_papers_in_archive":14,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/video-question-answering-on-msrvtt-mc","task":"Video Question Answering","dataset_variant":"MSRVTT-MC","rows":7,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"VIOLETv2","paper":"/paper/an-empirical-study-of-end-to-end-video","metrics":{"Accuracy":"97.6"},"code_links":[{"title":"tsujuifu/pytorch_empirical-mvm","url":"https://github.com/tsujuifu/pytorch_empirical-mvm"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/video-question-answering-on-msr-vtt-mc","task":"Video Question Answering","dataset_variant":"MSR-VTT-MC","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"ATP (1<-16)","paper":"/paper/revisiting-the-video-in-video-language","metrics":{"Accuracy":"93.2"},"code_links":[{"title":"stanfordvl/atp-video-language","url":"https://github.com/stanfordvl/atp-video-language"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/multi-granularity-correspondence-learning-1","title":"Multi-granularity Correspondence Learning from Long-term Noisy Videos","date":"2024-01-30","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/hitea-hierarchical-temporal-aware-video","title":"HiTeA: Hierarchical Temporal-Aware Video-Language Pre-training","date":"2022-12-30","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/vindlu-a-recipe-for-effective-video-and","title":"VindLU: A Recipe for Effective Video-and-Language Pretraining","date":"2022-12-09","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/an-empirical-study-of-end-to-end-video","title":"An Empirical Study of End-to-End Video-Language Transformers with Masked Visual Modeling","date":"2022-09-04","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/clover-towards-a-unified-video-language","title":"Clover: Towards A Unified Video-Language Alignment and Fusion Model","date":"2022-07-16","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/revealing-single-frame-bias-for-video-and","title":"Revealing Single Frame Bias for Video-and-Language Learning","date":"2022-06-07","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":4,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/revisiting-the-video-in-video-language","title":"Revisiting the \"Video\" in Video-Language Understanding","date":"2022-06-03","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":3,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":3,"samples_harvested":20,"samples_ran":9,"samples_unverified":11,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}