{"url":"/dataset/hybridqa","name":"HybridQA","full_name":null,"description_markdown":"A new large-scale question-answering dataset that requires reasoning on heterogeneous information. Each question is aligned with a Wikipedia table and multiple free-form corpora linked with the entities in the table. The questions are designed to aggregate both tabular information and text information, i.e., lack of either form would render the question unanswerable. \r\n\r\nSource: [HybridQA: A Dataset of Multi-Hop Question Answering over Tabular and Textual Data](/paper/hybridqa-a-dataset-of-multi-hop-question)","description_withheld":null,"homepage":"https://github.com/wenhuchen/HybridQA","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/hybridqa-a-dataset-of-multi-hop-question","title":"HybridQA: A Dataset of Multi-Hop Question Answering over Tabular and Textual Data","first_author":"Wenhu Chen","url":null},"license":null,"modalities":[],"tasks":[{"name":"Question Answering","url":"/task/question-answering","datasets_with_task":"/datasets/task/question-answering"},{"name":"Question Generation","url":"/task/question-generation","datasets_with_task":"/datasets/task/question-generation"},{"name":"Text-To-SQL","url":"/task/text-to-sql","datasets_with_task":"/datasets/task/text-to-sql"}],"languages":[],"variants":["HybridQA"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/wenhu/hybrid_qa","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/hybrid_qa","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/wenhuchen/HybridQA","url":"https://github.com/wenhuchen/HybridQA","frameworks":["pytorch"]}],"num_papers_in_archive":70,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/question-answering-on-hybridqa","task":"Question Answering","dataset_variant":"HybridQA","rows":4,"metrics":["ANS-EM"],"first_row_in_archive_order":{"model":"MAFiD","paper":"/paper/mafid-moving-average-equipped-fusion-in","metrics":{"ANS-EM":"65.4"},"code_links":[{"title":"ZIZUN/MAFiD","url":"https://github.com/ZIZUN/MAFiD"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/mafid-moving-average-equipped-fusion-in","title":"MAFiD: Moving Average Equipped Fusion-in-Decoder for Question Answering over Tabular and Textual Data","date":"2023-05-02","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/mate-multi-view-attention-for-table","title":"MATE: Multi-view Attention for Table Transformer Efficiency","date":"2021-09-09","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/end-to-end-multihop-retrieval-for","title":"Iterative Hierarchical Attention for Answering Complex Questions over Long Documents","date":"2021-06-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/hybridqa-a-dataset-of-multi-hop-question","title":"HybridQA: A Dataset of Multi-Hop Question Answering over Tabular and Textual Data","date":"2020-04-15","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":5,"samples_unverified":10,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":15,"samples_ran":5,"samples_unverified":10,"pointer_only_for_licence":2,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}