{"url":"/dataset/grailqa","name":"GrailQA","full_name":"Strongly Generalizable Question Answering","description_markdown":"GrailQA is a new large-scale, high-quality dataset for question answering on knowledge bases (KBQA) on Freebase with 64,331 questions annotated with both answers and corresponding logical forms in different syntax (i.e., SPARQL, S-expression, etc.). It can be used to test three levels of generalization in KBQA: i.i.d., compositional, and zero-shot.","description_withheld":null,"homepage":"https://dki-lab.github.io/GrailQA/","introduced_date":"2020-11-16","introduced_date_note":null,"introduced_by":{"paper":"/paper/beyond-i-i-d-three-levels-of-generalization","title":"Beyond I.I.D.: Three Levels of Generalization for Question Answering on Knowledge Bases","first_author":"Yu Gu","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Question Generation","url":"/task/question-generation","datasets_with_task":"/datasets/task/question-generation"},{"name":"Knowledge Base Question Answering","url":"/task/knowledge-base-question-answering","datasets_with_task":"/datasets/task/knowledge-base-question-answering"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["GrailQA","GrailQA-IID","GrailQA-Zero-Shot","GrailQA-Compositional"],"data_loaders":[],"num_papers_in_archive":34,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/question-generation-on-grailqa-compositional","task":"Question Generation","dataset_variant":"GrailQA-Compositional","rows":4,"metrics":["METEOR","BLEU","FactSpotter"],"first_row_in_archive_order":{"model":"FactJointGT","paper":"/paper/factspotter-evaluating-the-factual","metrics":{"BLEU":"31.25","FactSpotter":"97.10","METEOR":"36.21"},"code_links":[{"title":"guihuzhang/FactSpotter","url":"https://github.com/guihuzhang/FactSpotter"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/question-generation-on-grailqa-iid","task":"Question Generation","dataset_variant":"GrailQA-IID","rows":4,"metrics":["BLEU","METEOR","FactSpotter"],"first_row_in_archive_order":{"model":"FactT5B","paper":"/paper/factspotter-evaluating-the-factual","metrics":{"BLEU":"46.10","FactSpotter":"99.50","METEOR":"42.67"},"code_links":[{"title":"guihuzhang/FactSpotter","url":"https://github.com/guihuzhang/FactSpotter"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/question-generation-on-grailqa-zero-shot","task":"Question Generation","dataset_variant":"GrailQA-Zero-Shot","rows":4,"metrics":["METEOR","bleu","FactSpotter"],"first_row_in_archive_order":{"model":"JointGT","paper":"/paper/jointgt-graph-text-joint-representation","metrics":{"FactSpotter":"94.15","METEOR":"37.69","bleu":"32.94"},"code_links":[{"title":"thu-coai/JointGT","url":"https://github.com/thu-coai/JointGT"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/knowledge-base-question-answering-on-grailqa","task":"Knowledge Base Question Answering","dataset_variant":"GrailQA","rows":2,"metrics":["Compositional EM","Compositional F1","EM","I.I.D. EM","I.I.D. F1","Overall EM","Overall F1","Zero-shot EM","Zero-shot F1"],"first_row_in_archive_order":{"model":"ReTraCk","paper":"/paper/retrack-a-flexible-and-efficient-framework","metrics":{"Compositional EM":"61.5","Compositional F1":"70.9","I.I.D. EM":"84.4","I.I.D. F1":"87.5","Overall EM":"58.1","Overall F1":"65.3","Zero-shot EM":"44.6","Zero-shot F1":"52.5"},"code_links":[{"title":"microsoft/KC","url":"https://github.com/microsoft/KC/tree/main/papers/ReTraCk"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/paths-over-graph-knowledge-graph-enpowered","title":"Paths-over-Graph: Knowledge Graph Empowered Large Language Model Reasoning","date":"2024-10-18","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":1,"samples_unverified":11,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/factspotter-evaluating-the-factual","title":"FactSpotter: Evaluating the Factual Faithfulness of Graph-to-Text Generation","date":"2023-10-25","rows_on_this_dataset":6,"code_links":1,"syntology":null},{"paper":"/paper/retrack-a-flexible-and-efficient-framework","title":"ReTraCk: A Flexible and Efficient Framework for Knowledge Base Question Answering","date":"2021-08-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/jointgt-graph-text-joint-representation","title":"JointGT: Graph-Text Joint Representation Learning for Text Generation from Knowledge Graphs","date":"2021-06-19","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/investigating-pretrained-language-models-for","title":"Investigating Pretrained Language Models for Graph-to-Text Generation","date":"2020-07-16","rows_on_this_dataset":3,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":2,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":2,"samples_harvested":24,"samples_ran":3,"samples_unverified":21,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}