{"url":"/dataset/quasar-1","name":"QUASAR","full_name":"QUestion Answering by Search And Reading","description_markdown":"The Question Answering by Search And Reading (**QUASAR**) is a large-scale dataset consisting of [QUASAR-S](quasar-s) and [QUASAR-T](quasar-t). Each of these datasets is built to focus on evaluating systems devised to understand a natural language query, a large corpus of texts and to extract an answer to the question from the corpus. Specifically, QUASAR-S comprises 37,012 fill-in-the-gaps questions that are collected from the popular website Stack Overflow using entity tags. The QUASAR-T dataset contains 43,012 open-domain questions collected from various internet sources. The candidate documents for each question in this dataset are retrieved from an Apache Lucene based search engine built on top of the ClueWeb09 dataset.\r\n\r\nSource: [MRNN: A Multi-Resolution Neural Network with Duplex Attention for Document Retrieval in the Context of Question Answering](https://arxiv.org/abs/1911.00964)\r\nImage Source: [https://arxiv.org/pdf/1707.03904.pdf](https://arxiv.org/pdf/1707.03904.pdf)","description_withheld":null,"homepage":"https://github.com/bdhingra/quasar","introduced_date":"2017-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/quasar-datasets-for-question-answering-by","title":"Quasar: Datasets for Question Answering by Search and Reading","first_author":"Bhuwan Dhingra","url":null},"license":{"name":"BSD 2-Clause License","url":"https://github.com/bdhingra/quasar/blob/master/LICENSE"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Question Answering","url":"/task/question-answering","datasets_with_task":"/datasets/task/question-answering"},{"name":"Reading Comprehension","url":"/task/reading-comprehension","datasets_with_task":"/datasets/task/reading-comprehension"},{"name":"Open-Domain Question Answering","url":"/task/open-domain-question-answering","datasets_with_task":"/datasets/task/open-domain-question-answering"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["Quasar","QUASAR"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/sagnikrayc/quasar","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/bdhingra/quasar","url":"https://github.com/bdhingra/quasar","frameworks":[]}],"num_papers_in_archive":47,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/open-domain-question-answering-on-quasar","task":"Open-Domain Question Answering","dataset_variant":"Quasar","rows":6,"metrics":["EM (Quasar-T)","F1 (Quasar-T)"],"first_row_in_archive_order":{"model":"Evidence Aggregation via R^3 Re-Ranking","paper":"/paper/evidence-aggregation-for-answer-re-ranking-in","metrics":{"EM (Quasar-T)":"42.3","F1 (Quasar-T)":"49.6"},"code_links":[{"title":"shuohangwang/mprc","url":"https://github.com/shuohangwang/mprc"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/densely-connected-attention-propagation-for","title":"Densely Connected Attention Propagation for Reading Comprehension","date":"2018-11-10","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/denoising-distantly-supervised-open-domain","title":"Denoising Distantly Supervised Open-Domain Question Answering","date":"2018-07-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/evidence-aggregation-for-answer-re-ranking-in","title":"Evidence Aggregation for Answer Re-Ranking in Open-Domain Question Answering","date":"2017-11-14","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":7,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/r3-reinforced-reader-ranker-for-open-domain","title":"R$^3$: Reinforced Reader-Ranker for Open-Domain Question Answering","date":"2017-08-31","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/bidirectional-attention-flow-for-machine","title":"Bidirectional Attention Flow for Machine Comprehension","date":"2016-11-05","rows_on_this_dataset":1,"code_links":27,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":8,"samples_unverified":3,"pointer_only_for_licence":7,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/gated-attention-readers-for-text","title":"Gated-Attention Readers for Text Comprehension","date":"2016-06-05","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":1,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":3,"samples_harvested":22,"samples_ran":16,"samples_unverified":6,"pointer_only_for_licence":7,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}