{"url":"/dataset/tqa","name":"TQA","full_name":"Textbook Question Answering","description_markdown":"The TextbookQuestionAnswering (TQA) dataset is drawn from middle school science curricula. It consists of 1,076 lessons from Life Science, Earth Science and Physical Science textbooks. This includes 26,260 questions, including 12,567 that have an accompanying diagram.\r\n\r\nThe TQA dataset encourages work on the task of Multi-Modal Machine Comprehension (M3C) task. The M3C task builds on the popular Visual Question Answering (VQA) and Machine Comprehension (MC) paradigms by framing question answering as a machine comprehension task, where the context needed to answer questions is provided and composed of both text and images. The dataset constructed to showcase this task has been built from a middle school science curriculum that pairs a given question to a limited span of knowledge needed to answer it.\r\n\r\nSource: [Allen Institute for AI](https://allenai.org/data/tqa)","description_withheld":null,"homepage":"https://allenai.org/data/tqa","introduced_date":"2017-07-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/are-you-smarter-than-a-sixth-grader-textbook","title":"Are You Smarter Than a Sixth Grader? Textbook Question Answering for Multimodal Machine Comprehension","first_author":"Aniruddha Kembhavi","url":null},"license":{"name":"CC BY-SA 4.0","url":"https://creativecommons.org/licenses/by-sa/4.0/"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Question Answering","url":"/task/question-answering","datasets_with_task":"/datasets/task/question-answering"},{"name":"Visual Question Answering (VQA)","url":"/task/visual-question-answering","datasets_with_task":"/datasets/task/visual-question-answering"},{"name":"Reading Comprehension","url":"/task/reading-comprehension","datasets_with_task":"/datasets/task/reading-comprehension"},{"name":"Open-Domain Question Answering","url":"/task/open-domain-question-answering","datasets_with_task":"/datasets/task/open-domain-question-answering"}],"languages":[],"variants":["TQA"],"data_loaders":[],"num_papers_in_archive":48,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/open-domain-question-answering-on-tqa","task":"Open-Domain Question Answering","dataset_variant":"TQA","rows":2,"metrics":["Exact Match"],"first_row_in_archive_order":{"model":"UniK-QA","paper":"/paper/unified-open-domain-question-answering-with","metrics":{"Exact Match":"65.5"},"code_links":[{"title":"facebookresearch/UniK-QA","url":"https://github.com/facebookresearch/UniK-QA"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/efficient-passage-retrieval-with-hashing-for","title":"Efficient Passage Retrieval with Hashing for Open-domain Question Answering","date":"2021-06-02","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/unified-open-domain-question-answering-with","title":"UniK-QA: Unified Representations of Structured and Unstructured Knowledge for Open-Domain Question Answering","date":"2020-12-29","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}