{"url":"/dataset/bc5cdr","name":"BC5CDR","full_name":"BioCreative V CDR corpus","description_markdown":"**BC5CDR** corpus consists of 1500 PubMed articles with 4409 annotated chemicals, 5818 diseases and 3116 chemical-disease interactions.\r\n\r\nSource: [https://www.ncbi.nlm.nih.gov/research/bionlp/Data/](https://www.ncbi.nlm.nih.gov/research/bionlp/Data/)\r\nImage Source: [https://arxiv.org/pdf/1805.10586.pdf](https://arxiv.org/pdf/1805.10586.pdf)","description_withheld":null,"homepage":"https://github.com/JHnlp/BioCreative-V-CDR-Corpus","introduced_date":"2016-01-01","introduced_date_note":null,"introduced_by":{"paper":null,"title":"BioCreative V CDR task corpus: a resource for chemical disease relation extraction","first_author":null,"url":"https://doi.org/10.1093/database/baw068"},"license":{"name":"Custom","url":"https://biocreative.bioinformatics.udel.edu/tasks/biocreative-v/track-3-cdr/"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Named Entity Recognition (NER)","url":"/task/named-entity-recognition-ner","datasets_with_task":"/datasets/task/named-entity-recognition-ner"},{"name":"UIE","url":"/task/uie","datasets_with_task":"/datasets/task/uie"},{"name":"Weakly-Supervised Named Entity Recognition","url":"/task/weakly-supervised-named-entity-recognition","datasets_with_task":"/datasets/task/weakly-supervised-named-entity-recognition"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["BC5CDR","BC5CDR-disease","BC5CDR-chemical"],"data_loaders":[{"repo":"https://github.com/sq1balaji/EasyOCR-ChatBot","url":"https://github.com/sq1balaji/EasyOCR-ChatBot","frameworks":["tf","pytorch"]}],"num_papers_in_archive":191,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/named-entity-recognition-ner-on-bc5cdr","task":"Named Entity Recognition (NER)","dataset_variant":"BC5CDR","rows":16,"metrics":["F1"],"first_row_in_archive_order":{"model":"BINDER","paper":"/paper/optimizing-bi-encoder-for-named-entity","metrics":{"F1":"91.9"},"code_links":[{"title":"microsoft/binder","url":"https://github.com/microsoft/binder"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/named-entity-recognition-on-bc5cdr-chemical","task":"Named Entity Recognition (NER)","dataset_variant":"BC5CDR-chemical","rows":13,"metrics":["F1"],"first_row_in_archive_order":{"model":"Spark NLP","paper":"/paper/biomedical-named-entity-recognition-at-scale","metrics":{"F1":"94.88"},"code_links":[{"title":"JohnSnowLabs/spark-nlp-workshop","url":"https://github.com/JohnSnowLabs/spark-nlp-workshop/blob/master/tutorials/Certification_Trainings/Healthcare/1.4.Biomedical_NER_SparkNLP_paper_reproduce.ipynb"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/named-entity-recognition-on-bc5cdr-disease","task":"Named Entity Recognition (NER)","dataset_variant":"BC5CDR-disease","rows":10,"metrics":["F1"],"first_row_in_archive_order":{"model":"BioMegatron","paper":"/paper/biomegatron-larger-biomedical-domain-language","metrics":{"F1":"88.5"},"code_links":[{"title":"NVIDIA/NeMo","url":"https://github.com/NVIDIA/NeMo"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/uie-on-bc5cdr","task":"UIE","dataset_variant":"BC5CDR","rows":1,"metrics":["F1 score"],"first_row_in_archive_order":{"model":"KnowCoder-7b-IE","paper":"/paper/knowcoder-coding-structured-knowledge-into","metrics":{"F1 score":"89.3"},"code_links":[{"title":"ICT-GoKnow/KnowCoder","url":"https://github.com/ICT-GoKnow/KnowCoder"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/knowcoder-coding-structured-knowledge-into","title":"KnowCoder: Coding Structured Knowledge into LLMs for Universal Information Extraction","date":"2024-03-12","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/gollie-annotation-guidelines-improve-zero","title":"GoLLIE: Annotation Guidelines improve Zero-Shot Information-Extraction","date":"2023-10-05","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":25,"samples_ran":18,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/universalner-targeted-distillation-from-large","title":"UniversalNER: Targeted Distillation from Large Language Models for Open Named Entity Recognition","date":"2023-08-07","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":5,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/enhancing-label-consistency-on-document-level","title":"Enhancing Label Consistency on Document-level Named Entity Recognition","date":"2022-10-24","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/on-the-effectiveness-of-compact-biomedical","title":"On the Effectiveness of Compact Biomedical Transformers","date":"2022-09-07","rows_on_this_dataset":8,"code_links":1,"syntology":null},{"paper":"/paper/optimizing-bi-encoder-for-named-entity","title":"Optimizing Bi-Encoder for Named Entity Recognition via Contrastive Learning","date":"2022-08-30","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":4,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/accurate-clinical-and-biomedical-named-entity","title":"Accurate clinical and biomedical Named entity recognition at scale","date":"2022-07-19","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/hero-gang-neural-model-for-named-entity","title":"Hero-Gang Neural Model For Named Entity Recognition","date":"2022-05-15","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/linkbert-pretraining-language-models-with","title":"LinkBERT: Pretraining Language Models with Document Links","date":"2022-03-29","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":14,"samples_ran":0,"samples_unverified":14,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/focusing-on-possible-named-entities-in-active","title":"Focusing on Potential Named Entities During Active Label Acquisition","date":"2021-11-06","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/scifive-a-text-to-text-transformer-model-for","title":"SciFive: a text-to-text transformer model for biomedical literature","date":"2021-05-28","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":0,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/improving-named-entity-recognition-by","title":"Improving Named Entity Recognition by External Context Retrieving and Cooperative Learning","date":"2021-05-08","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":0,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/improving-biomedical-pretrained-language","title":"Improving Biomedical Pretrained Language Models with Knowledge","date":"2021-04-21","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/electramed-a-new-pre-trained-language","title":"ELECTRAMed: a new pre-trained language representation model for biomedical NLP","date":"2021-04-19","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/a-robust-and-domain-adaptive-approach-for-low","title":"A Robust and Domain-Adaptive Approach for Low-Resource Named Entity Recognition","date":"2021-01-02","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/improving-biomedical-named-entity-recognition","title":"Improving Biomedical Named Entity Recognition with Syntactic Information","date":"2020-11-25","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/biomedical-named-entity-recognition-at-scale","title":"Biomedical Named Entity Recognition at Scale","date":"2020-11-12","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/biomegatron-larger-biomedical-domain-language","title":"BioMegatron: Larger Biomedical Domain Language Model","date":"2020-10-12","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/transfer-learning-in-biomedical-natural","title":"Transfer Learning in Biomedical Natural Language Processing: An Evaluation of BERT and ELMo on Ten Benchmarking Datasets","date":"2019-06-13","rows_on_this_dataset":2,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/an-attention-based-bilstm-crf-approach-to","title":"An attention-based BiLSTM-CRF approach to document-level chemical named entity recognition","date":"2019-04-15","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/scibert-pretrained-contextualized-embeddings","title":"SciBERT: A Pretrained Language Model for Scientific Text","date":"2019-03-26","rows_on_this_dataset":2,"code_links":6,"syntology":null},{"paper":"/paper/collabonet-collaboration-of-deep-neural","title":"CollaboNet: collaboration of deep neural networks for biomedical named entity recognition","date":"2018-09-21","rows_on_this_dataset":1,"code_links":2,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":7,"samples_harvested":66,"samples_ran":27,"samples_unverified":39,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":4,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}