{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/clevr-a-diagnostic-dataset-for-compositional","title":"CLEVR: A Diagnostic Dataset for Compositional Language and Elementary Visual Reasoning","arxiv_id":"1612.06890","date":"2016-12-20","proceeding":"CVPR 2017 7","authors":["Justin Johnson","Bharath Hariharan","Laurens van der Maaten","Li Fei-Fei","C. Lawrence Zitnick","Ross Girshick"],"abstract":"When building artificial intelligence systems that can reason and answer\nquestions about visual data, we need diagnostic tests to analyze our progress\nand discover shortcomings. Existing benchmarks for visual question answering\ncan help, but have strong biases that models can exploit to correctly answer\nquestions without reasoning. They also conflate multiple sources of error,\nmaking it hard to pinpoint model weaknesses. We present a diagnostic dataset\nthat tests a range of visual reasoning abilities. It contains minimal biases\nand has detailed annotations describing the kind of reasoning each question\nrequires. We use this dataset to analyze a variety of modern visual reasoning\nsystems, providing novel insights into their abilities and limitations.","url_abs":"http://arxiv.org/abs/1612.06890v1","url_pdf":"http://arxiv.org/pdf/1612.06890v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"clevr-a-diagnostic-dataset-for-compositional","repo_url":"https://github.com/AlexKuhnle/film","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"clevr-a-diagnostic-dataset-for-compositional","repo_url":"https://github.com/Lucas2012/ProbabilisticNeuralProgrammedNetwork","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"Apache-2.0"}},{"paper_slug":"clevr-a-diagnostic-dataset-for-compositional","repo_url":"https://github.com/ethanjperez/film","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"clevr-a-diagnostic-dataset-for-compositional","repo_url":"https://github.com/necla-ml/SNLI-VE","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":{"status":"ok","spdx":"BSD-3-Clause"}},{"paper_slug":"clevr-a-diagnostic-dataset-for-compositional","repo_url":"https://github.com/pliang279/multiviz","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}}],"tasks":[{"task_slug":"diagnostic","task_name":"Diagnostic"},{"task_slug":"question-answering","task_name":"Question Answering"},{"task_slug":"visual-question-answering-1","task_name":"Visual Question Answering"},{"task_slug":"visual-question-answering","task_name":"Visual Question Answering (VQA)"},{"task_slug":"visual-reasoning","task_name":"Visual Reasoning"}],"methods":[],"datasets_introduced":[{"slug":"clevr","name":"CLEVR","full_name":"Compositional Language and Elementary Visual Reasoning"}],"methods_introduced":[],"results":[],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=1612.06890","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1612.06890"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/ethanjperez/film","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/necla-ml/SNLI-VE","reach":{"status":"ok","spdx":"BSD-3-Clause"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/AlexKuhnle/film","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/pliang279/multiviz","reach":{"status":"ok","spdx":"MIT"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/Lucas2012/ProbabilisticNeuralProgrammedNetwork","reach":{"status":"ok","spdx":"Apache-2.0"}}],"summary":{"ran_draft_wrong":2,"unverified":3},"by_repo_kind":{"listed":{"samples":5,"ran":2,"repositories":3}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":4,"samples":[{"code_sha256_prefix":"d095fd5888d50fa5","entry":"init_rnn","repo":"ethanjperez/film","repo_kind":"listed","path":"vr/models/film_gen.py","file_url":"https://github.com/ethanjperez/film/blob/HEAD/vr/models/film_gen.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"mcp_get_code":{"code_sha256":"d095fd5888d50fa5"}},{"code_sha256_prefix":"829ba4926365475d","entry":"init_rnn","repo":"AlexKuhnle/film","repo_kind":"listed","path":"vr/models/film_gen.py","file_url":"https://github.com/AlexKuhnle/film/blob/HEAD/vr/models/film_gen.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"mcp_get_code":{"code_sha256":"829ba4926365475d"}},{"code_sha256_prefix":"bc376bffc47027c6","entry":"coord_map","repo":"ethanjperez/film","repo_kind":"listed","path":"vr/models/filmed_net.py","file_url":"https://github.com/ethanjperez/film/blob/HEAD/vr/models/filmed_net.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"mcp_get_code":{"code_sha256":"bc376bffc47027c6"}},{"code_sha256_prefix":"78cd2988e1168e96","entry":"coord_map","repo":"AlexKuhnle/film","repo_kind":"listed","path":"vr/models/filmed_net.py","file_url":"https://github.com/AlexKuhnle/film/blob/HEAD/vr/models/filmed_net.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"mcp_get_code":{"code_sha256":"78cd2988e1168e96"}},{"code_sha256_prefix":"873e6f6b3cd35fc3","entry":"load_config","repo":"Lucas2012/ProbabilisticNeuralProgrammedNetwork","repo_kind":"listed","path":"lib/config.py","file_url":"https://github.com/Lucas2012/ProbabilisticNeuralProgrammedNetwork/blob/HEAD/lib/config.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"873e6f6b3cd35fc3"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}