{"url":"/dataset/newsqa","name":"NewsQA","full_name":null,"description_markdown":"The **NewsQA** dataset is a crowd-sourced machine reading comprehension dataset of 120,000 question-answer pairs.\r\n\r\n* Documents are CNN news articles.\r\n* Questions are written by human users in natural language.\r\n* Answers may be multiword passages of the source text.\r\n* Questions may be unanswerable.\r\n* NewsQA is collected using a 3-stage, siloed process.\r\n* Questioners see only an article’s headline and highlights.\r\n* Answerers see the question and the full article, then select an answer passage.\r\n* Validators see the article, the question, and a set of answers that they rank.\r\n* NewsQA is more natural and more challenging than previous datasets.\r\n\r\nSource: [https://www.microsoft.com/en-us/research/project/newsqa-dataset/](https://www.microsoft.com/en-us/research/project/newsqa-dataset/)\r\nImage Source: [Trischler et al](https://arxiv.org/pdf/1611.09830v3.pdf)","description_withheld":null,"homepage":"https://www.microsoft.com/en-us/research/project/newsqa-dataset/","introduced_date":"2017-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/newsqa-a-machine-comprehension-dataset","title":"NewsQA: A Machine Comprehension Dataset","first_author":"Adam Trischler","url":null},"license":{"name":"Custom","url":"https://www.microsoft.com/en-us/research/project/newsqa-dataset/#!download"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Question Answering","url":"/task/question-answering","datasets_with_task":"/datasets/task/question-answering"},{"name":"Reading Comprehension","url":"/task/reading-comprehension","datasets_with_task":"/datasets/task/reading-comprehension"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["NewsQA"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/Maluuba/newsqa","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/newsqa","frameworks":["tf","pytorch","jax"]}],"num_papers_in_archive":272,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/question-answering-on-newsqa","task":"Question Answering","dataset_variant":"NewsQA","rows":18,"metrics":["EM","F1"],"first_row_in_archive_order":{"model":"OpenAI/o3-2025-01-31-high","paper":"/paper/o3-mini-vs-deepseek-r1-which-one-is-safer","metrics":{"EM":"92.52","F1":"93.13"},"code_links":[{"title":"trust4ai/astral","url":"https://github.com/trust4ai/astral"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/o3-mini-vs-deepseek-r1-which-one-is-safer","title":"o3-mini vs DeepSeek-R1: Which One is Safer?","date":"2025-01-30","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/deepseek-r1-incentivizing-reasoning","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","date":"2025-01-22","rows_on_this_dataset":1,"code_links":4,"syntology":null},{"paper":"/paper/sieve-general-purpose-data-filtering-system","title":"GPT-4o as the Gold Standard: A Scalable and General Purpose Approach to Filter Language Model Pretraining Data","date":"2024-10-03","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/claude-3-5-sonnet-model-card-addendum","title":"Claude 3.5 Sonnet Model Card Addendum","date":"2024-06-24","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/gemini-1-5-unlocking-multimodal-understanding","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","date":"2024-03-08","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/dyrex-dynamic-query-representation-for","title":"DyREx: Dynamic Query Representation for Extractive Question Answering","date":"2022-10-26","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/0-1-deep-neural-networks-via-block-coordinate","title":"0/1 Deep Neural Networks via Block Coordinate Descent","date":"2022-06-19","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/time-series-transformer-generative","title":"Time-series Transformer Generative Adversarial Networks","date":"2022-05-23","rows_on_this_dataset":1,"code_links":5,"syntology":null},{"paper":"/paper/linkbert-pretraining-language-models-with","title":"LinkBERT: Pretraining Language Models with Document Links","date":"2022-03-29","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":14,"samples_ran":0,"samples_unverified":14,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/xai-for-transformers-better-explanations","title":"XAI for Transformers: Better Explanations through Conservative Propagation","date":"2022-02-15","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":2,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/learning-to-generate-questions-by-learning-to","title":"Learning to Generate Questions by Learning to Recover Answer-containing Sentences","date":"2021-08-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/thinking-like-transformers-1","title":"Thinking Like Transformers","date":"2021-06-13","rows_on_this_dataset":1,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":18,"samples_ran":11,"samples_unverified":7,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/spanbert-improving-pre-training-by","title":"SpanBERT: Improving Pre-training by Representing and Predicting Spans","date":"2019-07-24","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":3,"samples_unverified":12,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/densely-connected-attention-propagation-for","title":"Densely Connected Attention Propagation for Reading Comprehension","date":"2018-11-10","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/efficient-and-robust-question-answering-from","title":"Efficient and Robust Question Answering from Minimal Context over Documents","date":"2018-05-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/a-question-focused-multi-factor-attention","title":"A Question-Focused Multi-Factor Attention Network for Question Answering","date":"2018-01-25","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/making-neural-qa-as-simple-as-possible-but","title":"Making Neural QA as Simple as Possible but not Simpler","date":"2017-03-14","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/deepsense-a-unified-deep-learning-framework","title":"DeepSense: A Unified Deep Learning Framework for Time-Series Mobile Sensing Data Processing","date":"2016-11-07","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":6,"samples_harvested":64,"samples_ran":18,"samples_unverified":46,"pointer_only_for_licence":12,"papers_with_no_sample_that_ran":2,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}