{"url":"/dataset/jnlpba","name":"JNLPBA","full_name":"JNLPBA","description_markdown":"**JNLPBA** is a biomedical dataset that comes from the GENIA version 3.02 corpus (Kim et al., 2003). It was created with a controlled search on MEDLINE. From this search 2,000 abstracts were selected and hand annotated according to a small taxonomy of 48 classes based on a chemical classification. 36 terminal classes were used to annotate the GENIA corpus.","description_withheld":null,"homepage":"http://www.geniaproject.org/shared-tasks/bionlp-jnlpba-shared-task-2004","introduced_date":null,"introduced_date_note":null,"introduced_by":null,"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Named Entity Recognition (NER)","url":"/task/named-entity-recognition-ner","datasets_with_task":"/datasets/task/named-entity-recognition-ner"},{"name":"Relation Extraction","url":"/task/relation-extraction","datasets_with_task":"/datasets/task/relation-extraction"},{"name":"Medical Named Entity Recognition","url":"/task/medical-named-entity-recognition","datasets_with_task":"/datasets/task/medical-named-entity-recognition"}],"languages":[],"variants":["JNLPBA"],"data_loaders":[{"repo":"https://github.com/hanaeFSDM/BioNER","url":"https://github.com/hanaeFSDM/BioNER","frameworks":["tf"]}],"num_papers_in_archive":20,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/named-entity-recognition-ner-on-jnlpba","task":"Named Entity Recognition (NER)","dataset_variant":"JNLPBA","rows":17,"metrics":["F1"],"first_row_in_archive_order":{"model":"KeBioLM","paper":"/paper/improving-biomedical-pretrained-language","metrics":{"F1":"82.0"},"code_links":[{"title":"GanjinZero/KeBioLM","url":"https://github.com/GanjinZero/KeBioLM"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/relation-extraction-on-jnlpba","task":"Relation Extraction","dataset_variant":"JNLPBA","rows":1,"metrics":["F1"],"first_row_in_archive_order":{"model":"SciBERT (SciVocab)","paper":"/paper/scibert-pretrained-contextualized-embeddings","metrics":{"F1":"76.09"},"code_links":[{"title":"allenai/scibert","url":"https://github.com/allenai/scibert"},{"title":"charles9n/bert-sklearn","url":"https://github.com/charles9n/bert-sklearn"},{"title":"tetsu9923/scireviewgen","url":"https://github.com/tetsu9923/scireviewgen"},{"title":"georgetown-cset/ai-relevant-papers","url":"https://github.com/georgetown-cset/ai-relevant-papers"},{"title":"kuldeep7688/BioMedicalBertNer","url":"https://github.com/kuldeep7688/BioMedicalBertNer"},{"title":"hoangcuongnguyen2001/scibert-for-technique-classification","url":"https://github.com/hoangcuongnguyen2001/scibert-for-technique-classification"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/on-the-effectiveness-of-compact-biomedical","title":"On the Effectiveness of Compact Biomedical Transformers","date":"2022-09-07","rows_on_this_dataset":4,"code_links":1,"syntology":null},{"paper":"/paper/optimizing-bi-encoder-for-named-entity","title":"Optimizing Bi-Encoder for Named Entity Recognition via Contrastive Learning","date":"2022-08-30","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":4,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/miner-improving-out-of-vocabulary-named-1","title":"MINER: Improving Out-of-Vocabulary Named Entity Recognition from an Information Theoretic Perspective","date":"2022-04-09","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/linkbert-pretraining-language-models-with","title":"LinkBERT: Pretraining Language Models with Document Links","date":"2022-03-29","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":14,"samples_ran":0,"samples_unverified":14,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/scifive-a-text-to-text-transformer-model-for","title":"SciFive: a text-to-text transformer model for biomedical literature","date":"2021-05-28","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":0,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/improving-biomedical-pretrained-language","title":"Improving Biomedical Pretrained Language Models with Knowledge","date":"2021-04-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/improving-biomedical-named-entity-recognition","title":"Improving Biomedical Named Entity Recognition with Syntactic Information","date":"2020-11-25","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/biomedical-named-entity-recognition-at-scale","title":"Biomedical Named Entity Recognition at Scale","date":"2020-11-12","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/domain-specific-language-model-pretraining","title":"Domain-Specific Language Model Pretraining for Biomedical Natural Language Processing","date":"2020-07-31","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/bioflair-pretrained-pooled-contextualized","title":"BioFLAIR: Pretrained Pooled Contextualized Embeddings for Biomedical Sequence Labeling Tasks","date":"2019-08-13","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/scibert-pretrained-contextualized-embeddings","title":"SciBERT: A Pretrained Language Model for Scientific Text","date":"2019-03-26","rows_on_this_dataset":2,"code_links":6,"syntology":null},{"paper":"/paper/biobert-a-pre-trained-biomedical-language","title":"BioBERT: a pre-trained biomedical language representation model for biomedical text mining","date":"2019-01-25","rows_on_this_dataset":1,"code_links":19,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":25,"samples_ran":4,"samples_unverified":21,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":5,"samples_harvested":50,"samples_ran":8,"samples_unverified":42,"pointer_only_for_licence":1,"papers_with_no_sample_that_ran":3,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}