{"url":"/dataset/species-800","name":"Species-800","full_name":"Species-800","description_markdown":"**Species-800** is a corpus for species entities, which is based on manually annotated abstracts. It comprises 800 PubMed abstracts that contain identified organism mentions. To increase the corpus taxonomic mention diversity the 800 abstracts were collected by selecting 100 abstracts from the following 8 categories: bacteriology, botany, entomology, medicine, mycology, protistology, virology and zoology. 800 has been annotated with a focus at the species level; however, higher taxa mentions (such as genera, families and orders) have also been considered.","description_withheld":null,"homepage":"https://species.jensenlab.org/","introduced_date":null,"introduced_date_note":null,"introduced_by":null,"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Named Entity Recognition (NER)","url":"/task/named-entity-recognition-ner","datasets_with_task":"/datasets/task/named-entity-recognition-ner"},{"name":"Medical Named Entity Recognition","url":"/task/medical-named-entity-recognition","datasets_with_task":"/datasets/task/medical-named-entity-recognition"}],"languages":[],"variants":["Species-800"],"data_loaders":[],"num_papers_in_archive":5,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/named-entity-recognition-on-species-800","task":"Named Entity Recognition (NER)","dataset_variant":"Species-800","rows":4,"metrics":["F1"],"first_row_in_archive_order":{"model":"BioBERT","paper":"/paper/biobert-a-pre-trained-biomedical-language","metrics":{"F1":"75.31"},"code_links":[{"title":"dmis-lab/biobert","url":"https://github.com/dmis-lab/biobert"},{"title":"EmilyAlsentzer/clinicalBERT","url":"https://github.com/EmilyAlsentzer/clinicalBERT"},{"title":"naver/biobert-pretrained","url":"https://github.com/naver/biobert-pretrained"},{"title":"plkmo/BERT-Relation-Extraction","url":"https://github.com/plkmo/BERT-Relation-Extraction"},{"title":"ncbi-nlp/NCBI_BERT","url":"https://github.com/ncbi-nlp/NCBI_BERT"},{"title":"re-search/DocProduct","url":"https://github.com/re-search/DocProduct"},{"title":"charles9n/bert-sklearn","url":"https://github.com/charles9n/bert-sklearn"},{"title":"dmis-lab/bern","url":"https://github.com/dmis-lab/bern"},{"title":"MeRajat/SolvingAlmostAnythingWithBert","url":"https://github.com/MeRajat/SolvingAlmostAnythingWithBert"},{"title":"phucdev/TL_Bio_RE","url":"https://github.com/phucdev/TL_Bio_RE"},{"title":"ManasRMohanty/DS5500-capstone","url":"https://github.com/ManasRMohanty/DS5500-capstone"},{"title":"kuldeep7688/BioMedicalBertNer","url":"https://github.com/kuldeep7688/BioMedicalBertNer"},{"title":"mocherson/aki_bert","url":"https://github.com/mocherson/aki_bert"},{"title":"jpablou/Matching-The-Blanks-Ths","url":"https://github.com/jpablou/Matching-The-Blanks-Ths"},{"title":"rahul-1996/KGraphs-QA","url":"https://github.com/rahul-1996/KGraphs-QA"},{"title":"ardakdemir/my_bert_ner","url":"https://github.com/ardakdemir/my_bert_ner"},{"title":"cypressd1999/FYP_2021","url":"https://github.com/cypressd1999/FYP_2021"},{"title":"hieudepchai/BERT_IE","url":"https://github.com/hieudepchai/BERT_IE"},{"title":"Dean/BioBERT-DAGsHub","url":"https://dagshub.com/Dean/BioBERT-DAGsHub"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/scifive-a-text-to-text-transformer-model-for","title":"SciFive: a text-to-text transformer model for biomedical literature","date":"2021-05-28","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":0,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/improving-biomedical-named-entity-recognition","title":"Improving Biomedical Named Entity Recognition with Syntactic Information","date":"2020-11-25","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/biomedical-named-entity-recognition-at-scale","title":"Biomedical Named Entity Recognition at Scale","date":"2020-11-12","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/biobert-a-pre-trained-biomedical-language","title":"BioBERT: a pre-trained biomedical language representation model for biomedical text mining","date":"2019-01-25","rows_on_this_dataset":1,"code_links":19,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":25,"samples_ran":4,"samples_unverified":21,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":2,"samples_harvested":30,"samples_ran":4,"samples_unverified":26,"pointer_only_for_licence":1,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}