{"url":"/dataset/bioasq","name":"BioASQ","full_name":"Biomedical Semantic Indexing and Question Answering","description_markdown":"**BioASQ** is a question answering dataset. Instances in the BioASQ dataset are composed of a question (Q), human-annotated answers (A), and the relevant contexts (C) (also called snippets).\r\n\r\nSource: [Transferability of Natural Language Inference to Biomedical Question Answering](https://arxiv.org/abs/2007.00217)\r\nImage Source: [http://participants-area.bioasq.org/datasets/](http://participants-area.bioasq.org/datasets/)","description_withheld":null,"homepage":"http://participants-area.bioasq.org/datasets/","introduced_date":"2015-01-01","introduced_date_note":null,"introduced_by":{"paper":null,"title":"An overview of the BIOASQ large-scale biomedical semantic indexing and question answering competition","first_author":null,"url":"https://doi.org/10.1186/s12859-015-0564-6"},"license":{"name":"Unknown","url":null},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Question Answering","url":"/task/question-answering","datasets_with_task":"/datasets/task/question-answering"},{"name":"Information Retrieval","url":"/task/information-retrieval","datasets_with_task":"/datasets/task/information-retrieval"},{"name":"Zero-shot Text Search","url":"/task/zero-shot-text-search","datasets_with_task":"/datasets/task/zero-shot-text-search"},{"name":"Word Embeddings","url":"/task/word-embeddings","datasets_with_task":"/datasets/task/word-embeddings"}],"languages":[],"variants":["BioASQ"],"data_loaders":[],"num_papers_in_archive":192,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/question-answering-on-bioasq","task":"Question Answering","dataset_variant":"BioASQ","rows":7,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"BioLinkBERT (large)","paper":"/paper/linkbert-pretraining-language-models-with","metrics":{"Accuracy":"94.8"},"code_links":[{"title":"michiyasunaga/LinkBERT","url":"https://github.com/michiyasunaga/LinkBERT"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/evaluation-of-large-language-model","title":"Evaluation of large language model performance on the Biomedical Language Understanding and Reasoning Benchmark","date":"2024-05-17","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/galactica-a-large-language-model-for-science-1","title":"Galactica: A Large Language Model for Science","date":"2022-11-16","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/linkbert-pretraining-language-models-with","title":"LinkBERT: Pretraining Language Models with Document Links","date":"2022-03-29","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":14,"samples_ran":0,"samples_unverified":14,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/domain-specific-language-model-pretraining","title":"Domain-Specific Language Model Pretraining for Biomedical Natural Language Processing","date":"2020-07-31","rows_on_this_dataset":1,"code_links":2,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":2,"samples_harvested":16,"samples_ran":0,"samples_unverified":16,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":2,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}