{"url":"/dataset/ncbi-disease-1","name":"NCBI Disease","full_name":null,"description_markdown":"The **NCBI Disease** corpus consists of 793 PubMed abstracts, which are separated into training (593), development (100) and test (100) subsets. The NCBI Disease corpus is annotated with disease mentions, using concept identifiers from either MeSH or OMIM.\r\n\r\nSource: [A Neural Multi-Task Learning Framework to Jointly Model Medical Named Entity Recognition and Normalization](https://arxiv.org/abs/1812.06081)","description_withheld":null,"homepage":"https://www.ncbi.nlm.nih.gov/CBBresearch/Dogan/DISEASE/","introduced_date":"2014-01-01","introduced_date_note":null,"introduced_by":{"paper":null,"title":"NCBI disease corpus: A resource for disease name recognition and concept normalization","first_author":null,"url":"http://dx.doi.org/10.1016/j.jbi.2013.12.006"},"license":{"name":"Unknown","url":null},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Named Entity Recognition (NER)","url":"/task/named-entity-recognition-ner","datasets_with_task":"/datasets/task/named-entity-recognition-ner"},{"name":"UIE","url":"/task/uie","datasets_with_task":"/datasets/task/uie"},{"name":"Named Entity Recognition","url":"/task/named-entity-recognition-1","datasets_with_task":"/datasets/task/named-entity-recognition-1"}],"languages":[],"variants":["NCBI-disease","NCBI Disease","ncbi_disease"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/ncbi/ncbi_disease","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/scillm/ncbi_disease","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/ncbi_disease","frameworks":["tf","pytorch","jax"]}],"num_papers_in_archive":154,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/named-entity-recognition-ner-on-ncbi-disease","task":"Named Entity Recognition (NER)","dataset_variant":"NCBI-disease","rows":26,"metrics":["F1"],"first_row_in_archive_order":{"model":"BioBERT","paper":"/paper/biobert-a-pre-trained-biomedical-language","metrics":{"F1":"89.71"},"code_links":[{"title":"dmis-lab/biobert","url":"https://github.com/dmis-lab/biobert"},{"title":"EmilyAlsentzer/clinicalBERT","url":"https://github.com/EmilyAlsentzer/clinicalBERT"},{"title":"naver/biobert-pretrained","url":"https://github.com/naver/biobert-pretrained"},{"title":"plkmo/BERT-Relation-Extraction","url":"https://github.com/plkmo/BERT-Relation-Extraction"},{"title":"ncbi-nlp/NCBI_BERT","url":"https://github.com/ncbi-nlp/NCBI_BERT"},{"title":"re-search/DocProduct","url":"https://github.com/re-search/DocProduct"},{"title":"charles9n/bert-sklearn","url":"https://github.com/charles9n/bert-sklearn"},{"title":"dmis-lab/bern","url":"https://github.com/dmis-lab/bern"},{"title":"MeRajat/SolvingAlmostAnythingWithBert","url":"https://github.com/MeRajat/SolvingAlmostAnythingWithBert"},{"title":"phucdev/TL_Bio_RE","url":"https://github.com/phucdev/TL_Bio_RE"},{"title":"ManasRMohanty/DS5500-capstone","url":"https://github.com/ManasRMohanty/DS5500-capstone"},{"title":"kuldeep7688/BioMedicalBertNer","url":"https://github.com/kuldeep7688/BioMedicalBertNer"},{"title":"mocherson/aki_bert","url":"https://github.com/mocherson/aki_bert"},{"title":"jpablou/Matching-The-Blanks-Ths","url":"https://github.com/jpablou/Matching-The-Blanks-Ths"},{"title":"rahul-1996/KGraphs-QA","url":"https://github.com/rahul-1996/KGraphs-QA"},{"title":"ardakdemir/my_bert_ner","url":"https://github.com/ardakdemir/my_bert_ner"},{"title":"cypressd1999/FYP_2021","url":"https://github.com/cypressd1999/FYP_2021"},{"title":"hieudepchai/BERT_IE","url":"https://github.com/hieudepchai/BERT_IE"},{"title":"Dean/BioBERT-DAGsHub","url":"https://dagshub.com/Dean/BioBERT-DAGsHub"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/named-entity-recognition-on-ncbi-disease","task":"Named Entity Recognition (NER)","dataset_variant":"NCBI Disease","rows":1,"metrics":["F1"],"first_row_in_archive_order":{"model":"UniNER-7B","paper":"/paper/universalner-targeted-distillation-from-large","metrics":{"F1":"86.96"},"code_links":[{"title":"universal-ner/universal-ner","url":"https://github.com/universal-ner/universal-ner"},{"title":"emma1066/retrieval-augmented-it-openner","url":"https://github.com/emma1066/retrieval-augmented-it-openner"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/uie-on-ncbi-disease","task":"UIE","dataset_variant":"ncbi_disease","rows":1,"metrics":["F1 score"],"first_row_in_archive_order":{"model":"KnowCoder-7b-IE","paper":"/paper/knowcoder-coding-structured-knowledge-into","metrics":{"F1 score":"83.8"},"code_links":[{"title":"ICT-GoKnow/KnowCoder","url":"https://github.com/ICT-GoKnow/KnowCoder"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/evaluation-of-large-language-model","title":"Evaluation of large language model performance on the Biomedical Language Understanding and Reasoning Benchmark","date":"2024-05-17","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/knowcoder-coding-structured-knowledge-into","title":"KnowCoder: Coding Structured Knowledge into LLMs for Universal Information Extraction","date":"2024-03-12","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/nuner-entity-recognition-encoder-pre-training","title":"NuNER: Entity Recognition Encoder Pre-training via LLM-Annotated Data","date":"2024-02-23","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/gollie-annotation-guidelines-improve-zero","title":"GoLLIE: Annotation Guidelines improve Zero-Shot Information-Extraction","date":"2023-10-05","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":25,"samples_ran":18,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/universalner-targeted-distillation-from-large","title":"UniversalNER: Targeted Distillation from Large Language Models for Open Named Entity Recognition","date":"2023-08-07","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":5,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/comparing-and-combining-some-popular-ner","title":"Comparing and combining some popular NER approaches on Biomedical tasks","date":"2023-05-30","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/on-the-effectiveness-of-compact-biomedical","title":"On the Effectiveness of Compact Biomedical Transformers","date":"2022-09-07","rows_on_this_dataset":4,"code_links":1,"syntology":null},{"paper":"/paper/linkbert-pretraining-language-models-with","title":"LinkBERT: Pretraining Language Models with Document Links","date":"2022-03-29","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":14,"samples_ran":0,"samples_unverified":14,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/bern2-an-advanced-neural-biomedical-named","title":"BERN2: an advanced neural biomedical named entity recognition and normalization tool","date":"2022-01-06","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/focusing-on-possible-named-entities-in-active","title":"Focusing on Potential Named Entities During Active Label Acquisition","date":"2021-11-06","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/scifive-a-text-to-text-transformer-model-for","title":"SciFive: a text-to-text transformer model for biomedical literature","date":"2021-05-28","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":0,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/improving-named-entity-recognition-by","title":"Improving Named Entity Recognition by External Context Retrieving and Cooperative Learning","date":"2021-05-08","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":0,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/improving-biomedical-pretrained-language","title":"Improving Biomedical Pretrained Language Models with Knowledge","date":"2021-04-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/electramed-a-new-pre-trained-language","title":"ELECTRAMed: a new pre-trained language representation model for biomedical NLP","date":"2021-04-19","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/a-robust-and-domain-adaptive-approach-for-low","title":"A Robust and Domain-Adaptive Approach for Low-Resource Named Entity Recognition","date":"2021-01-02","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/improving-biomedical-named-entity-recognition","title":"Improving Biomedical Named Entity Recognition with Syntactic Information","date":"2020-11-25","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/biomedical-named-entity-recognition-at-scale","title":"Biomedical Named Entity Recognition at Scale","date":"2020-11-12","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/biomegatron-larger-biomedical-domain-language","title":"BioMegatron: Larger Biomedical Domain Language Model","date":"2020-10-12","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/domain-specific-language-model-pretraining","title":"Domain-Specific Language Model Pretraining for Biomedical Natural Language Processing","date":"2020-07-31","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/learning-a-unified-named-entity-tagger-from","title":"Learning A Unified Named Entity Tagger From Multiple Partially Annotated Corpora For Efficient Adaptation","date":"2019-09-25","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/a-neural-named-entity-recognition-and-multi","title":"A Neural Named Entity Recognition and Multi-Type Normalization Tool for Biomedical Text Mining","date":"2019-06-04","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/scibert-pretrained-contextualized-embeddings","title":"SciBERT: A Pretrained Language Model for Scientific Text","date":"2019-03-26","rows_on_this_dataset":2,"code_links":6,"syntology":null},{"paper":"/paper/biobert-a-pre-trained-biomedical-language","title":"BioBERT: a pre-trained biomedical language representation model for biomedical text mining","date":"2019-01-25","rows_on_this_dataset":1,"code_links":19,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":25,"samples_ran":4,"samples_unverified":21,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":7,"samples_harvested":87,"samples_ran":29,"samples_unverified":58,"pointer_only_for_licence":1,"papers_with_no_sample_that_ran":3,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}