{"url":"/dataset/bc4chemd","name":"BC4CHEMD","full_name":"BioCreative IV Chemical compound and drug name recognition","description_markdown":"Introduced by Krallinger et al. in [The CHEMDNER corpus of chemicals and drugs and its annotation principles](https://jcheminf.biomedcentral.com/articles/10.1186/1758-2946-7-S1-S2)\r\n\r\n**BC4CHEMD** is a collection of 10,000 PubMed abstracts that contain a total of 84,355 chemical entity mentions labeled manually by expert chemistry literature curators.","description_withheld":null,"homepage":"https://biocreative.bioinformatics.udel.edu/resources/biocreative-iv/chemdner-corpus/","introduced_date":null,"introduced_date_note":null,"introduced_by":null,"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Named Entity Recognition (NER)","url":"/task/named-entity-recognition-ner","datasets_with_task":"/datasets/task/named-entity-recognition-ner"},{"name":"NER","url":"/task/cg","datasets_with_task":"/datasets/task/cg"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["BC4CHEMD"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/drAbreu/bc4chemd_ner","frameworks":["tf","pytorch","jax"]}],"num_papers_in_archive":6,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/named-entity-recognition-on-bc4chemd","task":"Named Entity Recognition (NER)","dataset_variant":"BC4CHEMD","rows":7,"metrics":["F1"],"first_row_in_archive_order":{"model":"BertForTokenClassification (Spark NLP)","paper":"/paper/accurate-clinical-and-biomedical-named-entity","metrics":{"F1":"94.39"},"code_links":[{"title":"JohnSnowLabs/spark-nlp-workshop","url":"https://github.com/JohnSnowLabs/spark-nlp-workshop"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/universalner-targeted-distillation-from-large","title":"UniversalNER: Targeted Distillation from Large Language Models for Open Named Entity Recognition","date":"2023-08-07","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":12,"samples_ran":7,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/accurate-clinical-and-biomedical-named-entity","title":"Accurate clinical and biomedical Named entity recognition at scale","date":"2022-07-19","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/bern2-an-advanced-neural-biomedical-named","title":"BERN2: an advanced neural biomedical named entity recognition and normalization tool","date":"2022-01-06","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/biomedical-named-entity-recognition-at-scale","title":"Biomedical Named Entity Recognition at Scale","date":"2020-11-12","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/a-neural-named-entity-recognition-and-multi","title":"A Neural Named Entity Recognition and Multi-Type Normalization Tool for Biomedical Text Mining","date":"2019-06-04","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/an-attention-based-bilstm-crf-approach-to","title":"An attention-based BiLSTM-CRF approach to document-level chemical named entity recognition","date":"2019-04-15","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":2,"samples_harvested":14,"samples_ran":9,"samples_unverified":5,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}