{"url":"/dataset/vqa-cp","name":"VQA-CP","full_name":null,"description_markdown":"The **VQA-CP** dataset was constructed by reorganizing VQA v2 such that the correlation between the question type and correct answer differs in the training and test splits. For example, the most common answer to questions starting with What sport… is tennis in the training set, but skiing in the test set. A model that guesses an answer primarily from the question will perform poorly.\n\nSource: [Unshuffling Data for Improved Generalization](https://arxiv.org/abs/2002.11894)\nImage Source: [https://arxiv.org/pdf/1712.00377.pdf](https://arxiv.org/pdf/1712.00377.pdf)","description_withheld":null,"homepage":"https://www.cc.gatech.edu/~aagrawal307/vqa-cp/","introduced_date":"2018-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/dont-just-assume-look-and-answer-overcoming","title":"Don't Just Assume; Look and Answer: Overcoming Priors for Visual Question Answering","first_author":"Aishwarya Agrawal","url":null},"license":null,"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Visual Question Answering (VQA)","url":"/task/visual-question-answering","datasets_with_task":"/datasets/task/visual-question-answering"}],"languages":[],"variants":["VQA-CP"],"data_loaders":[],"num_papers_in_archive":13,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/visual-question-answering-on-vqa-cp","task":"Visual Question Answering (VQA)","dataset_variant":"VQA-CP","rows":10,"metrics":["Score"],"first_row_in_archive_order":{"model":"CSS","paper":"/paper/counterfactual-samples-synthesizing-for","metrics":{"Score":"58.95"},"code_links":[{"title":"yanxinzju/CSS-VQA","url":"https://github.com/yanxinzju/CSS-VQA"},{"title":"FengSuSky/CCB-VQA","url":"https://github.com/FengSuSky/CCB-VQA"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/greedy-gradient-ensemble-for-robust-visual","title":"Greedy Gradient Ensemble for Robust Visual Question Answering","date":"2021-07-27","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/removing-bias-in-multi-modal-classifiers","title":"Removing Bias in Multi-modal Classifiers: Regularization by Maximizing Functional Entropies","date":"2020-10-21","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/counterfactual-samples-synthesizing-for","title":"Counterfactual Samples Synthesizing for Robust Visual Question Answering","date":"2020-03-14","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/dont-take-the-easy-way-out-ensemble-based","title":"Don't Take the Easy Way Out: Ensemble Based Methods for Avoiding Known Dataset Biases","date":"2019-09-09","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":3,"samples_unverified":3,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/learning-by-abstraction-the-neural-state","title":"Learning by Abstraction: The Neural State Machine","date":"2019-07-09","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":21,"samples_ran":3,"samples_unverified":18,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/rubi-reducing-unimodal-biases-in-visual","title":"RUBi: Reducing Unimodal Biases in Visual Question Answering","date":"2019-06-24","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/self-critical-reasoning-for-robust-visual","title":"Self-Critical Reasoning for Robust Visual Question Answering","date":"2019-05-24","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/murel-multimodal-relational-reasoning-for","title":"MUREL: Multimodal Relational Reasoning for Visual Question Answering","date":"2019-02-25","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/learning-visual-question-answering-by","title":"Learning Visual Question Answering by Bootstrapping Hard Attention","date":"2018-08-01","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":5,"samples_harvested":33,"samples_ran":10,"samples_unverified":23,"pointer_only_for_licence":9,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}