{"url":"/dataset/prontoqa","name":"PrOntoQA","full_name":"Proof and Ontology-Generated Question-Answering","description_markdown":"**PrOntoQA** is a question-answering dataset which generates examples with chains-of-thought that describe the reasoning required to answer the questions correctly. The sentences in the examples are syntactically simple and amenable to semantic parsing. It can be used to formally analyze the predicted chain-of-thought from large language models such as GPT-3.\r\n\r\nSource: [Language Models Are Greedy Reasoners: A Systematic Formal Analysis of Chain-of-Thought](/paper/language-models-are-greedy-reasoners-a)\r\n\r\nImage Source: [https://arxiv.org/pdf/2210.01240v1.pdf](https://arxiv.org/pdf/2210.01240v1.pdf)","description_withheld":null,"homepage":"https://github.com/asaparov/prontoqa","introduced_date":"2022-10-03","introduced_date_note":null,"introduced_by":{"paper":"/paper/language-models-are-greedy-reasoners-a","title":"Language Models Are Greedy Reasoners: A Systematic Formal Analysis of Chain-of-Thought","first_author":"Abulhair Saparov","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Question Answering","url":"/task/question-answering","datasets_with_task":"/datasets/task/question-answering"},{"name":"Mathematical Reasoning","url":"/task/mathematical-reasoning","datasets_with_task":"/datasets/task/mathematical-reasoning"}],"languages":[],"variants":["PrOntoQA"],"data_loaders":[],"num_papers_in_archive":55,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}