{"url":"/dataset/medmentions","name":"MedMentions","full_name":null,"description_markdown":"MedMentions is a new manually annotated resource for the recognition of biomedical concepts. What distinguishes MedMentions from other annotated biomedical corpora is its size (over 4,000 abstracts and over 350,000 linked mentions), as well as the size of the concept ontology (over 3 million concepts from UMLS 2017) and its broad coverage of biomedical disciplines. \r\n\r\nSource: [MedMentions: A Large Biomedical Corpus Annotated with UMLS Concepts](/paper/medmentions-a-large-biomedical-corpus)","description_withheld":null,"homepage":"https://github.com/chanzuckerberg/MedMentions","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/medmentions-a-large-biomedical-corpus","title":"MedMentions: A Large Biomedical Corpus Annotated with UMLS Concepts","first_author":"Sunil Mohan","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"},{"name":"Medical","url":"/datasets/modality/medical"}],"tasks":[{"name":"Named Entity Recognition (NER)","url":"/task/named-entity-recognition-ner","datasets_with_task":"/datasets/task/named-entity-recognition-ner"},{"name":"Entity Linking","url":"/task/entity-linking","datasets_with_task":"/datasets/task/entity-linking"},{"name":"Entity Extraction using GAN","url":"/task/entity-extraction","datasets_with_task":"/datasets/task/entity-extraction"}],"languages":[],"variants":["MedMentions"],"data_loaders":[{"repo":"https://github.com/chanzuckerberg/MedMentions","url":"https://github.com/chanzuckerberg/MedMentions","frameworks":[]}],"num_papers_in_archive":48,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/entity-linking-on-medmentions","task":"Entity Linking","dataset_variant":"MedMentions","rows":3,"metrics":["Accuracy","Recall@64"],"first_row_in_archive_order":{"model":"ArboEL","paper":"/paper/entity-linking-via-explicit-mention-mention-1","metrics":{"Accuracy":"75.73"},"code_links":[{"title":"dhdhagar/arboEL","url":"https://github.com/dhdhagar/arboEL"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/entity-linking-via-explicit-mention-mention-1","title":"Entity Linking via Explicit Mention-Mention Coreference Modeling","date":"2022-07-01","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/biobart-pretraining-and-evaluation-of-a","title":"BioBART: Pretraining and Evaluation of A Biomedical Generative Language Model","date":"2022-04-08","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}