{"url":"/dataset/chemprot","name":"ChemProt","full_name":null,"description_markdown":"**ChemProt** consists of 1,820 PubMed abstracts with chemical-protein interactions annotated by domain experts and was used in the BioCreative VI text mining chemical-protein interactions shared task.\r\n\r\nSource: [Peng et al.](https://arxiv.org/pdf/1906.05474v2.pdf)","description_withheld":null,"homepage":"https://biocreative.bioinformatics.udel.edu/news/corpora/chemprot-corpus-biocreative-vi/","introduced_date":null,"introduced_date_note":null,"introduced_by":null,"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"},{"name":"Biomedical","url":"/datasets/modality/biomedical"}],"tasks":[{"name":"Relation Extraction","url":"/task/relation-extraction","datasets_with_task":"/datasets/task/relation-extraction"}],"languages":[],"variants":["ChemProt"],"data_loaders":[],"num_papers_in_archive":16,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/relation-extraction-on-chemprot","task":"Relation Extraction","dataset_variant":"ChemProt","rows":13,"metrics":["F1","Micro F1"],"first_row_in_archive_order":{"model":"SciBert (Finetune)","paper":"/paper/scibert-pretrained-contextualized-embeddings","metrics":{"F1":"83.64"},"code_links":[{"title":"allenai/scibert","url":"https://github.com/allenai/scibert"},{"title":"charles9n/bert-sklearn","url":"https://github.com/charles9n/bert-sklearn"},{"title":"tetsu9923/scireviewgen","url":"https://github.com/tetsu9923/scireviewgen"},{"title":"georgetown-cset/ai-relevant-papers","url":"https://github.com/georgetown-cset/ai-relevant-papers"},{"title":"kuldeep7688/BioMedicalBertNer","url":"https://github.com/kuldeep7688/BioMedicalBertNer"},{"title":"hoangcuongnguyen2001/scibert-for-technique-classification","url":"https://github.com/hoangcuongnguyen2001/scibert-for-technique-classification"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/linkbert-pretraining-language-models-with","title":"LinkBERT: Pretraining Language Models with Document Links","date":"2022-03-29","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":14,"samples_ran":0,"samples_unverified":14,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/biom-transformers-building-large-biomedical","title":"BioM-Transformers: Building Large Biomedical Language Models with BERT, ALBERT and ELECTRA","date":"2021-06-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/scifive-a-text-to-text-transformer-model-for","title":"SciFive: a text-to-text transformer model for biomedical literature","date":"2021-05-28","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":0,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/improving-biomedical-pretrained-language","title":"Improving Biomedical Pretrained Language Models with Knowledge","date":"2021-04-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/electramed-a-new-pre-trained-language","title":"ELECTRAMed: a new pre-trained language representation model for biomedical NLP","date":"2021-04-19","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/characterbert-reconciling-elmo-and-bert-for","title":"CharacterBERT: Reconciling ELMo and BERT for Word-Level Open-Vocabulary Representations From Characters","date":"2020-10-20","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":5,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/biomegatron-larger-biomedical-domain-language","title":"BioMegatron: Larger Biomedical Domain Language Model","date":"2020-10-12","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/domain-specific-language-model-pretraining","title":"Domain-Specific Language Model Pretraining for Biomedical Natural Language Processing","date":"2020-07-31","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/transfer-learning-in-biomedical-natural","title":"Transfer Learning in Biomedical Natural Language Processing: An Evaluation of BERT and ELMo on Ten Benchmarking Datasets","date":"2019-06-13","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/scibert-pretrained-contextualized-embeddings","title":"SciBERT: A Pretrained Language Model for Scientific Text","date":"2019-03-26","rows_on_this_dataset":2,"code_links":6,"syntology":null},{"paper":"/paper/biobert-a-pre-trained-biomedical-language","title":"BioBERT: a pre-trained biomedical language representation model for biomedical text mining","date":"2019-01-25","rows_on_this_dataset":1,"code_links":19,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":25,"samples_ran":4,"samples_unverified":21,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":5,"samples_harvested":54,"samples_ran":9,"samples_unverified":45,"pointer_only_for_licence":1,"papers_with_no_sample_that_ran":3,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}