{"url":"/dataset/bc2gm","name":"BC2GM","full_name":"BC2GM","description_markdown":"Created by Smith et al. at 2008, the BioCreative II Gene Mention Recognition (BC2GM) Dataset contains data where participants are asked to identify a gene mention in a sentence by giving its start and end characters. The training set consists of a set of sentences, and for each sentence a set of gene mentions (GENE annotations). [registration required for access], in English language. Containing 20 in n/a file format.","description_withheld":null,"homepage":"https://biocreative.bioinformatics.udel.edu/tasks/","introduced_date":null,"introduced_date_note":null,"introduced_by":null,"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Named Entity Recognition (NER)","url":"/task/named-entity-recognition-ner","datasets_with_task":"/datasets/task/named-entity-recognition-ner"},{"name":"UIE","url":"/task/uie","datasets_with_task":"/datasets/task/uie"},{"name":"Medical Named Entity Recognition","url":"/task/medical-named-entity-recognition","datasets_with_task":"/datasets/task/medical-named-entity-recognition"}],"languages":[],"variants":["BC2GM"],"data_loaders":[],"num_papers_in_archive":13,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/named-entity-recognition-on-bc2gm","task":"Named Entity Recognition (NER)","dataset_variant":"BC2GM","rows":13,"metrics":["F1"],"first_row_in_archive_order":{"model":"Spark NLP","paper":"/paper/biomedical-named-entity-recognition-at-scale","metrics":{"F1":"88.75"},"code_links":[{"title":"JohnSnowLabs/spark-nlp-workshop","url":"https://github.com/JohnSnowLabs/spark-nlp-workshop/blob/master/tutorials/Certification_Trainings/Healthcare/1.4.Biomedical_NER_SparkNLP_paper_reproduce.ipynb"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/uie-on-bc2gm","task":"UIE","dataset_variant":"BC2GM","rows":1,"metrics":["F1 score"],"first_row_in_archive_order":{"model":"KnowCoder-7b-IE","paper":"/paper/knowcoder-coding-structured-knowledge-into","metrics":{"F1 score":"82.0"},"code_links":[{"title":"ICT-GoKnow/KnowCoder","url":"https://github.com/ICT-GoKnow/KnowCoder"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/knowcoder-coding-structured-knowledge-into","title":"KnowCoder: Coding Structured Knowledge into LLMs for Universal Information Extraction","date":"2024-03-12","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/universalner-targeted-distillation-from-large","title":"UniversalNER: Targeted Distillation from Large Language Models for Open Named Entity Recognition","date":"2023-08-07","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":5,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/on-the-effectiveness-of-compact-biomedical","title":"On the Effectiveness of Compact Biomedical Transformers","date":"2022-09-07","rows_on_this_dataset":4,"code_links":1,"syntology":null},{"paper":"/paper/hero-gang-neural-model-for-named-entity","title":"Hero-Gang Neural Model For Named Entity Recognition","date":"2022-05-15","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/linkbert-pretraining-language-models-with","title":"LinkBERT: Pretraining Language Models with Document Links","date":"2022-03-29","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":14,"samples_ran":0,"samples_unverified":14,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/bern2-an-advanced-neural-biomedical-named","title":"BERN2: an advanced neural biomedical named entity recognition and normalization tool","date":"2022-01-06","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/improving-biomedical-pretrained-language","title":"Improving Biomedical Pretrained Language Models with Knowledge","date":"2021-04-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/improving-biomedical-named-entity-recognition","title":"Improving Biomedical Named Entity Recognition with Syntactic Information","date":"2020-11-25","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/biomedical-named-entity-recognition-at-scale","title":"Biomedical Named Entity Recognition at Scale","date":"2020-11-12","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/domain-specific-language-model-pretraining","title":"Domain-Specific Language Model Pretraining for Biomedical Natural Language Processing","date":"2020-07-31","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/a-neural-named-entity-recognition-and-multi","title":"A Neural Named Entity Recognition and Multi-Type Normalization Tool for Biomedical Text Mining","date":"2019-06-04","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":3,"samples_harvested":28,"samples_ran":7,"samples_unverified":21,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}