{"url":"/dataset/searchqa","name":"SearchQA","full_name":null,"description_markdown":"SearchQA was built using an in-production, commercial search engine. It closely reflects the full pipeline of a (hypothetical) general question-answering system, which consists of information retrieval and answer synthesis. \r\n\r\nSource: [SearchQA: A New Q&A Dataset Augmented with Context from a Search Engine](https://arxiv.org/pdf/1704.05179.pdf)","description_withheld":null,"homepage":"https://github.com/nyu-dl/dl4ir-searchQA","introduced_date":"2017-04-18","introduced_date_note":null,"introduced_by":{"paper":"/paper/searchqa-a-new-qa-dataset-augmented-with","title":"SearchQA: A New Q&A Dataset Augmented with Context from a Search Engine","first_author":"Matthew Dunn","url":null},"license":{"name":"Unknown","url":null},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Open-Domain Question Answering","url":"/task/open-domain-question-answering","datasets_with_task":"/datasets/task/open-domain-question-answering"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["SearchQA"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/kyunghyuncho/search_qa","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/search_qa","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/nyu-dl/dl4ir-searchQA","url":"https://github.com/nyu-dl/dl4ir-searchQA","frameworks":["pytorch"]}],"num_papers_in_archive":133,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/open-domain-question-answering-on-searchqa","task":"Open-Domain Question Answering","dataset_variant":"SearchQA","rows":14,"metrics":["EM","N-gram F1","Unigram Acc","F1"],"first_row_in_archive_order":{"model":"Cluster-Former (#C=512)","paper":"/paper/cluster-former-clustering-based-sparse","metrics":{"EM":"68.0"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/cluster-former-clustering-based-sparse","title":"Cluster-Former: Clustering-based Sparse Transformer for Long-Range Dependency Encoding","date":"2020-09-13","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/reformer-the-efficient-transformer-1","title":"Reformer: The Efficient Transformer","date":"2020-01-13","rows_on_this_dataset":1,"code_links":10,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":6,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/multi-passage-bert-a-globally-normalized-bert","title":"Multi-passage BERT: A Globally Normalized BERT Model for Open-domain Question Answering","date":"2019-08-22","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/spanbert-improving-pre-training-by","title":"SpanBERT: Improving Pre-training by Representing and Predicting Spans","date":"2019-07-24","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":3,"samples_unverified":12,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/190410509","title":"Generating Long Sequences with Sparse Transformers","date":"2019-04-23","rows_on_this_dataset":1,"code_links":7,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":5,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/densely-connected-attention-propagation-for","title":"Densely Connected Attention Propagation for Reading Comprehension","date":"2018-11-10","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/multi-granular-sequence-encoding-via-dilated","title":"Multi-Granular Sequence Encoding via Dilated Compositional Units for Reading Comprehension","date":"2018-10-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/denoising-distantly-supervised-open-domain","title":"Denoising Distantly Supervised Open-Domain Question Answering","date":"2018-07-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/focused-hierarchical-rnns-for-conditional","title":"Focused Hierarchical RNNs for Conditional Sequence Processing","date":"2018-06-12","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/a-question-focused-multi-factor-attention","title":"A Question-Focused Multi-Factor Attention Network for Question Answering","date":"2018-01-25","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/r3-reinforced-reader-ranker-for-open-domain","title":"R$^3$: Reinforced Reader-Ranker for Open-Domain Question Answering","date":"2017-08-31","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/reading-wikipedia-to-answer-open-domain","title":"Reading Wikipedia to Answer Open-Domain Questions","date":"2017-03-31","rows_on_this_dataset":1,"code_links":10,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/text-understanding-with-the-attention-sum","title":"Text Understanding with the Attention Sum Reader Network","date":"2016-03-04","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":0,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":5,"samples_harvested":40,"samples_ran":15,"samples_unverified":25,"pointer_only_for_licence":5,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}