{"url":"/dataset/qasper","name":"QASPER","full_name":null,"description_markdown":"**QASPER** is a dataset for question answering on scientific research papers. It consists of 5,049 questions over 1,585 Natural Language Processing papers. Each question is written by an NLP practitioner who read only the title and abstract of the corresponding paper, and the question seeks information present in the full text. The questions are then answered by a separate set of NLP practitioners who also provide supporting evidence to answers.","description_withheld":null,"homepage":"https://allenai.org/data/qasper","introduced_date":"2021-05-07","introduced_date_note":null,"introduced_by":{"paper":"/paper/a-dataset-of-information-seeking-questions","title":"A Dataset of Information-Seeking Questions and Answers Anchored in Research Papers","first_author":"Pradeep Dasigi","url":null},"license":{"name":"CC BY 4.0","url":"https://creativecommons.org/licenses/by/4.0/"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Question Answering","url":"/task/question-answering","datasets_with_task":"/datasets/task/question-answering"},{"name":"Evidence Selection","url":"/task/evidence-selection","datasets_with_task":"/datasets/task/evidence-selection"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["QASPER"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/scillm/qasper","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/allenai/qasper","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/qasper","frameworks":["tf","pytorch","jax"]}],"num_papers_in_archive":102,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/question-answering-on-qasper","task":"Question Answering","dataset_variant":"QASPER","rows":1,"metrics":["Token F1"],"first_row_in_archive_order":{"model":"Longformer Encoder Decoder (base)","paper":"/paper/a-dataset-of-information-seeking-questions","metrics":{"Token F1":"33.63"},"code_links":[{"title":"allenai/qasper-led-baseline","url":"https://github.com/allenai/qasper-led-baseline"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/a-dataset-of-information-seeking-questions","title":"A Dataset of Information-Seeking Questions and Answers Anchored in Research Papers","date":"2021-05-07","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":1,"samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}