{"url":"/dataset/dureader","name":"DuReader","full_name":null,"description_markdown":"**DuReader** is a large-scale open-domain Chinese machine reading comprehension dataset. The dataset consists of 200K questions, 420K answers and 1M documents. The questions and documents are based on Baidu Search and Baidu Zhidao. The answers are manually generated. The dataset additionally provides question type annotations – each question was manually annotated as either Entity, Description or YesNo and one of Fact or Opinion.\r\n\r\nSource: [https://arxiv.org/pdf/1711.05073v4.pdf](https://arxiv.org/pdf/1711.05073v4.pdf)\r\nImage Source: [https://arxiv.org/pdf/1711.05073v4.pdf](https://arxiv.org/pdf/1711.05073v4.pdf)","description_withheld":null,"homepage":"http://ai.baidu.com/broad/subordinate?dataset=dureader","introduced_date":"2018-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/dureader-a-chinese-machine-reading","title":"DuReader: a Chinese Machine Reading Comprehension Dataset from Real-world Applications","first_author":"Wei He","url":null},"license":{"name":"Unknown","url":null},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Reading Comprehension","url":"/task/reading-comprehension","datasets_with_task":"/datasets/task/reading-comprehension"},{"name":"Open-Domain Question Answering","url":"/task/open-domain-question-answering","datasets_with_task":"/datasets/task/open-domain-question-answering"},{"name":"Reading Comprehension (Zero-Shot)","url":"/task/reading-comprehension-zero-shot","datasets_with_task":"/datasets/task/reading-comprehension-zero-shot"},{"name":"Reading Comprehension (One-Shot)","url":"/task/reading-comprehension-one-shot","datasets_with_task":"/datasets/task/reading-comprehension-one-shot"},{"name":"Reading Comprehension (Few-Shot)","url":"/task/reading-comprehension-few-shot","datasets_with_task":"/datasets/task/reading-comprehension-few-shot"}],"languages":[{"name":"Chinese","url":"/datasets/language/chinese"}],"variants":["DuReader"],"data_loaders":[],"num_papers_in_archive":65,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/open-domain-question-answering-on-dureader","task":"Open-Domain Question Answering","dataset_variant":"DuReader","rows":2,"metrics":["EM"],"first_row_in_archive_order":{"model":"ERNIE 2.0 Large","paper":"/paper/ernie-20-a-continual-pre-training-framework","metrics":{"EM":"64.2"},"code_links":[{"title":"PaddlePaddle/PaddleNLP","url":"https://github.com/PaddlePaddle/PaddleNLP/tree/develop/model_zoo/ernie-1.0"},{"title":"PaddlePaddle/ERNIE","url":"https://github.com/PaddlePaddle/ERNIE"},{"title":"DataScienceNigeria/ERNIE-2.0-from-Baidu-Inc.","url":"https://github.com/DataScienceNigeria/ERNIE-2.0-from-Baidu-Inc."}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/ernie-20-a-continual-pre-training-framework","title":"ERNIE 2.0: A Continual Pre-training Framework for Language Understanding","date":"2019-07-29","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":1,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}