{"url":"/dataset/linnaeus","name":"LINNAEUS","full_name":null,"description_markdown":"LINNAEUS is a general-purpose dictionary matching software, capable of processing multiple types of document formats in the biomedical domain (MEDLINE, PMC, BMC, OTMI, text, etc.). It can produce multiple types of output (XML, HTML, tab-separated-value file, or save to a database). It also contains methods for acting as a server (including load balancing across several servers), allowing clients to request matching over a network. A package with files for recognizing and identifying species names is available for LINNAEUS, showing 94% recall and 97% precision compared to LINNAEUS-species-corpus.\r\n\r\nSource: [LINNAEUS](http://linnaeus.sourceforge.net/)","description_withheld":null,"homepage":"http://linnaeus.sourceforge.net/","introduced_date":null,"introduced_date_note":null,"introduced_by":null,"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Named Entity Recognition (NER)","url":"/task/named-entity-recognition-ner","datasets_with_task":"/datasets/task/named-entity-recognition-ner"}],"languages":[],"variants":["LINNAEUS"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/cambridgeltl/linnaeus","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/linnaeus","frameworks":["tf","pytorch","jax"]}],"num_papers_in_archive":5,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/named-entity-recognition-on-linnaeus","task":"Named Entity Recognition (NER)","dataset_variant":"LINNAEUS","rows":6,"metrics":["F1"],"first_row_in_archive_order":{"model":"BLSTM-CNN-Char (SparkNLP)","paper":"/paper/biomedical-named-entity-recognition-at-scale","metrics":{"F1":"86.26"},"code_links":[{"title":"JohnSnowLabs/spark-nlp-workshop","url":"https://github.com/JohnSnowLabs/spark-nlp-workshop/blob/master/tutorials/Certification_Trainings/Healthcare/1.4.Biomedical_NER_SparkNLP_paper_reproduce.ipynb"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/bern2-an-advanced-neural-biomedical-named","title":"BERN2: an advanced neural biomedical named entity recognition and normalization tool","date":"2022-01-06","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/biomedical-named-entity-recognition-at-scale","title":"Biomedical Named Entity Recognition at Scale","date":"2020-11-12","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/bioflair-pretrained-pooled-contextualized","title":"BioFLAIR: Pretrained Pooled Contextualized Embeddings for Biomedical Sequence Labeling Tasks","date":"2019-08-13","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/a-neural-named-entity-recognition-and-multi","title":"A Neural Named Entity Recognition and Multi-Type Normalization Tool for Biomedical Text Mining","date":"2019-06-04","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":1,"samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}