{"url":"/dataset/vcr","name":"VCR","full_name":"Visual Commonsense Reasoning","description_markdown":"**Visual Commonsense Reasoning** (**VCR**) is a large-scale dataset for cognition-level visual understanding. Given a challenging question about an image, machines need to present two sub-tasks: answer correctly and provide a rationale justifying its answer. The VCR dataset contains over 212K (training), 26K (validation) and 25K (testing) questions, answers and rationales derived from 110K movie scenes.\r\n\r\nSource: [Visual Commonsense R-CNN](https://arxiv.org/abs/2002.12204)\r\nImage Source: [From Recognition to Cognition: Visual Commonsense Reasoning](https://paperswithcode.com/paper/from-recognition-to-cognition-visual/)","description_withheld":null,"homepage":"https://visualcommonsense.com/","introduced_date":"2019-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/from-recognition-to-cognition-visual","title":"From Recognition to Cognition: Visual Commonsense Reasoning","first_author":"Rowan Zellers","url":null},"license":{"name":"Custom","url":"https://visualcommonsense.com/license/"},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Visual Question Answering (VQA)","url":"/task/visual-question-answering","datasets_with_task":"/datasets/task/visual-question-answering"},{"name":"Visual Commonsense Reasoning","url":"/task/visual-commonsense-reasoning","datasets_with_task":"/datasets/task/visual-commonsense-reasoning"},{"name":"Explanation Generation","url":"/task/explanation-generation","datasets_with_task":"/datasets/task/explanation-generation"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["VCR","VCR (QA-R) test","VCR (QA-R) dev","VCR (Q-AR) test","VCR (Q-AR) dev","VCR (Q-A) test","VCR (Q-A) dev"],"data_loaders":[],"num_papers_in_archive":179,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/visual-question-answering-on-vcr-q-a-test","task":"Visual Question Answering (VQA)","dataset_variant":"VCR (Q-A) test","rows":11,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"GPT4RoI","paper":"/paper/gpt4roi-instruction-tuning-large-language","metrics":{"Accuracy":"89.4"},"code_links":[{"title":"jshilong/gpt4roi","url":"https://github.com/jshilong/gpt4roi"},{"title":"sunsmarterjie/chatterbox","url":"https://github.com/sunsmarterjie/chatterbox"},{"title":"qiujihao19/artemis","url":"https://github.com/qiujihao19/artemis"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-question-answering-on-vcr-qa-r-test","task":"Visual Question Answering (VQA)","dataset_variant":"VCR (QA-R) test","rows":8,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"GPT4RoI","paper":"/paper/gpt4roi-instruction-tuning-large-language","metrics":{"Accuracy":"91.0"},"code_links":[{"title":"jshilong/gpt4roi","url":"https://github.com/jshilong/gpt4roi"},{"title":"sunsmarterjie/chatterbox","url":"https://github.com/sunsmarterjie/chatterbox"},{"title":"qiujihao19/artemis","url":"https://github.com/qiujihao19/artemis"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-question-answering-on-vcr-q-ar-test","task":"Visual Question Answering (VQA)","dataset_variant":"VCR (Q-AR) test","rows":7,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"GPT4RoI","paper":"/paper/gpt4roi-instruction-tuning-large-language","metrics":{"Accuracy":"81.6"},"code_links":[{"title":"jshilong/gpt4roi","url":"https://github.com/jshilong/gpt4roi"},{"title":"sunsmarterjie/chatterbox","url":"https://github.com/sunsmarterjie/chatterbox"},{"title":"qiujihao19/artemis","url":"https://github.com/qiujihao19/artemis"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-question-answering-on-vcr-q-a-dev","task":"Visual Question Answering (VQA)","dataset_variant":"VCR (Q-A) dev","rows":3,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"VL-BERTLARGE","paper":"/paper/vl-bert-pre-training-of-generic-visual","metrics":{"Accuracy":"75.5"},"code_links":[{"title":"jackroos/VL-BERT","url":"https://github.com/jackroos/VL-BERT"},{"title":"ImperialNLP/BertGen","url":"https://github.com/ImperialNLP/BertGen"},{"title":"jules-samaran/vl-bert","url":"https://github.com/jules-samaran/vl-bert"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-question-answering-on-vcr-q-ar-dev","task":"Visual Question Answering (VQA)","dataset_variant":"VCR (Q-AR) dev","rows":3,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"VL-BERTLARGE","paper":"/paper/vl-bert-pre-training-of-generic-visual","metrics":{"Accuracy":"58.9"},"code_links":[{"title":"jackroos/VL-BERT","url":"https://github.com/jackroos/VL-BERT"},{"title":"ImperialNLP/BertGen","url":"https://github.com/ImperialNLP/BertGen"},{"title":"jules-samaran/vl-bert","url":"https://github.com/jules-samaran/vl-bert"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-question-answering-on-vcr-qa-r-dev","task":"Visual Question Answering (VQA)","dataset_variant":"VCR (QA-R) dev","rows":3,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"VL-BERTLARGE","paper":"/paper/vl-bert-pre-training-of-generic-visual","metrics":{"Accuracy":"77.9"},"code_links":[{"title":"jackroos/VL-BERT","url":"https://github.com/jackroos/VL-BERT"},{"title":"ImperialNLP/BertGen","url":"https://github.com/ImperialNLP/BertGen"},{"title":"jules-samaran/vl-bert","url":"https://github.com/jules-samaran/vl-bert"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/explanation-generation-on-vcr","task":"Explanation Generation","dataset_variant":"VCR","rows":2,"metrics":["Human Explanation Rating"],"first_row_in_archive_order":{"model":"OFA-X-MT","paper":"/paper/harnessing-the-power-of-multi-task","metrics":{"Human Explanation Rating":"77.3"},"code_links":[{"title":"ofa-x/ofa-x","url":"https://github.com/ofa-x/ofa-x"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-commonsense-reasoning-on-vcr-q-a-dev","task":"Visual Commonsense Reasoning","dataset_variant":"VCR (Q-A) dev","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"PEVL","paper":"/paper/pevl-position-enhanced-pre-training-and","metrics":{"Accuracy":"75.1"},"code_links":[{"title":"thunlp/pevl","url":"https://github.com/thunlp/pevl"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-commonsense-reasoning-on-vcr-q-a-test","task":"Visual Commonsense Reasoning","dataset_variant":"VCR (Q-A) test","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"PEVL","paper":"/paper/pevl-position-enhanced-pre-training-and","metrics":{"Accuracy":"76.0"},"code_links":[{"title":"thunlp/pevl","url":"https://github.com/thunlp/pevl"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-commonsense-reasoning-on-vcr-q-ar-dev","task":"Visual Commonsense Reasoning","dataset_variant":"VCR (Q-AR) dev","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"PEVL","paper":"/paper/pevl-position-enhanced-pre-training-and","metrics":{"Accuracy":"57.8"},"code_links":[{"title":"thunlp/pevl","url":"https://github.com/thunlp/pevl"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-commonsense-reasoning-on-vcr-q-ar-test","task":"Visual Commonsense Reasoning","dataset_variant":"VCR (Q-AR) test","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"PEVL","paper":"/paper/pevl-position-enhanced-pre-training-and","metrics":{"Accuracy":"58.6"},"code_links":[{"title":"thunlp/pevl","url":"https://github.com/thunlp/pevl"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-commonsense-reasoning-on-vcr-qa-r-dev","task":"Visual Commonsense Reasoning","dataset_variant":"VCR (QA-R) dev","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"PEVL","paper":"/paper/pevl-position-enhanced-pre-training-and","metrics":{"Accuracy":"76.4"},"code_links":[{"title":"thunlp/pevl","url":"https://github.com/thunlp/pevl"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-commonsense-reasoning-on-vcr-qa-r-test","task":"Visual Commonsense Reasoning","dataset_variant":"VCR (QA-R) test","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"PEVL","paper":"/paper/pevl-position-enhanced-pre-training-and","metrics":{"Accuracy":"76.7"},"code_links":[{"title":"thunlp/pevl","url":"https://github.com/thunlp/pevl"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/gpt4roi-instruction-tuning-large-language","title":"GPT4RoI: Instruction Tuning Large Language Model on Region-of-Interest","date":"2023-07-07","rows_on_this_dataset":3,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":3,"samples_unverified":2,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/harnessing-the-power-of-multi-task","title":"Harnessing the Power of Multi-Task Pretraining for Ground-Truth Level Natural Language Explanations","date":"2022-12-08","rows_on_this_dataset":4,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":1,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/pevl-position-enhanced-pre-training-and","title":"PEVL: Position-enhanced Pre-training and Prompt Tuning for Vision-language Models","date":"2022-05-23","rows_on_this_dataset":6,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":3,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/multimodal-adaptive-distillation-for","title":"Multimodal Adaptive Distillation for Leveraging Unimodal Encoders for Vision-Language Tasks","date":"2022-04-22","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/unifying-vision-and-language-tasks-via-text","title":"Unifying Vision-and-Language Tasks via Text Generation","date":"2021-02-04","rows_on_this_dataset":3,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":2,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/kvl-bert-knowledge-enhanced-visual-and","title":"KVL-BERT: Knowledge Enhanced Visual-and-Linguistic BERT for Visual Commonsense Reasoning","date":"2020-12-13","rows_on_this_dataset":3,"code_links":0,"syntology":null},{"paper":"/paper/ernie-vil-knowledge-enhanced-vision-language","title":"ERNIE-ViL: Knowledge Enhanced Vision-Language Representations Through Scene Graph","date":"2020-06-30","rows_on_this_dataset":3,"code_links":0,"syntology":null},{"paper":"/paper/uniter-learning-universal-image-text-1","title":"UNITER: UNiversal Image-TExt Representation Learning","date":"2019-09-25","rows_on_this_dataset":5,"code_links":7,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vl-bert-pre-training-of-generic-visual","title":"VL-BERT: Pre-training of Generic Visual-Linguistic Representations","date":"2019-08-22","rows_on_this_dataset":9,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/visualbert-a-simple-and-performant-baseline","title":"VisualBERT: A Simple and Performant Baseline for Vision and Language","date":"2019-08-09","rows_on_this_dataset":6,"code_links":10,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":4,"samples_unverified":5,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":7,"samples_harvested":45,"samples_ran":16,"samples_unverified":29,"pointer_only_for_licence":13,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}