{"url":"/dataset/hoc-1","name":"HOC","full_name":"Hallmarks of Cancer","description_markdown":"The **Hallmarks of Cancer** (**HOC*) corpus consists of 1852 PubMed publication abstracts manually annotated by experts according to the Hallmarks of Cancer taxonomy. The taxonomy consists of 37 classes in a hierarchy. Zero or more class labels are assigned to each sentence in the corpus. \r\n\r\nSource: [Hallmarks of Cancer Corpus](https://s-baker.net/resource/hoc/)\r\n\r\nImage source: [Hallmarks of Cancer Corpus](https://s-baker.net/resource/hoc/)","description_withheld":null,"homepage":"https://s-baker.net/resource/hoc/","introduced_date":"2015-10-09","introduced_date_note":null,"introduced_by":{"paper":"/paper/automatic-semantic-classification-of","title":"Automatic semantic classification of scientific literature according to the hallmarks of cancer","first_author":"Simon Baker","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Document Classification","url":"/task/document-classification","datasets_with_task":"/datasets/task/document-classification"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["HOC"],"data_loaders":[],"num_papers_in_archive":37,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/document-classification-on-hoc","task":"Document Classification","dataset_variant":"HOC","rows":5,"metrics":["F1","Micro F1"],"first_row_in_archive_order":{"model":"BioLinkBERT (large)","paper":"/paper/linkbert-pretraining-language-models-with","metrics":{"F1":"88.1","Micro F1":"84.87"},"code_links":[{"title":"michiyasunaga/LinkBERT","url":"https://github.com/michiyasunaga/LinkBERT"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/biogpt-generative-pre-trained-transformer-for","title":"BioGPT: Generative Pre-trained Transformer for Biomedical Text Generation and Mining","date":"2022-10-19","rows_on_this_dataset":1,"code_links":4,"syntology":null},{"paper":"/paper/linkbert-pretraining-language-models-with","title":"LinkBERT: Pretraining Language Models with Document Links","date":"2022-03-29","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":14,"samples_ran":0,"samples_unverified":14,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/scifive-a-text-to-text-transformer-model-for","title":"SciFive: a text-to-text transformer model for biomedical literature","date":"2021-05-28","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":0,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/domain-specific-language-model-pretraining","title":"Domain-Specific Language Model Pretraining for Biomedical Natural Language Processing","date":"2020-07-31","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/transfer-learning-in-biomedical-natural","title":"Transfer Learning in Biomedical Natural Language Processing: An Evaluation of BERT and ELMo on Ten Benchmarking Datasets","date":"2019-06-13","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":3,"samples_harvested":21,"samples_ran":0,"samples_unverified":21,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":3,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}