{"url":"/dataset/adversarialqa","name":"AdversarialQA","full_name":null,"description_markdown":"We have created three new Reading Comprehension datasets constructed using an adversarial model-in-the-loop.\r\n\r\nWe use three different models; BiDAF (Seo et al., 2016), BERTLarge (Devlin et al., 2018), and RoBERTaLarge (Liu et al., 2019) in the annotation loop and construct three datasets; D(BiDAF), D(BERT), and D(RoBERTa), each with 10,000 training examples, 1,000 validation, and 1,000 test examples.\r\n\r\nThe adversarial human annotation paradigm ensures that these datasets consist of questions that current state-of-the-art models (at least the ones used as adversaries in the annotation loop) find challenging. The three AdversarialQA round 1 datasets provide a training and evaluation resource for such methods.","description_withheld":null,"homepage":"https://adversarialqa.github.io/","introduced_date":"2020-02-02","introduced_date_note":null,"introduced_by":{"paper":"/paper/beat-the-ai-investigating-adversarial-human","title":"Beat the AI: Investigating Adversarial Human Annotation for Reading Comprehension","first_author":"Max Bartolo","url":null},"license":{"name":"CC BY-SA 3.0","url":"https://creativecommons.org/licenses/by-sa/3.0/"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Question Answering","url":"/task/question-answering","datasets_with_task":"/datasets/task/question-answering"},{"name":"Reading Comprehension","url":"/task/reading-comprehension","datasets_with_task":"/datasets/task/reading-comprehension"},{"name":"Machine Reading Comprehension","url":"/task/machine-reading-comprehension","datasets_with_task":"/datasets/task/machine-reading-comprehension"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["AdversarialQA","adversarial_qa"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/UCLNLP/adversarial_qa","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/adversarial_qa","frameworks":["tf","pytorch","jax"]}],"num_papers_in_archive":26,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/reading-comprehension-on-adversarialqa","task":"Reading Comprehension","dataset_variant":"AdversarialQA","rows":3,"metrics":["Overall: F1","D(BiDAF): F1","D(BERT): F1","D(RoBERTa): F1"],"first_row_in_archive_order":{"model":"RoBERTa-Large","paper":"/paper/beat-the-ai-investigating-adversarial-human","metrics":{"D(BERT): F1":"65.5","D(BiDAF): F1":"74.1","D(RoBERTa): F1":"53.4","Overall: F1":"64.4"},"code_links":[{"title":"maxbartolo/adversarialQA","url":"https://github.com/maxbartolo/adversarialQA"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/question-answering-on-adversarial-qa","task":"Question Answering","dataset_variant":"adversarial_qa","rows":0,"metrics":["Exact Match","F1"],"first_row_in_archive_order":null,"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/beat-the-ai-investigating-adversarial-human","title":"Beat the AI: Investigating Adversarial Human Annotation for Reading Comprehension","date":"2020-02-02","rows_on_this_dataset":3,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}