{"url":"/sota/document-classification-on-hoc","task":{"name":"Document Classification","url":"/task/document-classification","note":null},"dataset":{"name":"HOC","url":"/dataset/hoc-1"},"category":"Natural Language Processing","categories":["Natural Language Processing"],"category_note":null,"description":"**Document Classification** is a procedure of assigning one or more labels to a document from a predetermined set of labels.\n\n\n<span class=\"description-source\">Source: [Long-length Legal Document Classification ](https://arxiv.org/abs/1912.06905)</span>","description_from":"task","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","rank":"the archive's row order at snapshot; not re-ranked","rows_end_at":"2025-07-28","rows_withheld_as_spam":0,"metric_values":"the archive's strings, untouched"},"metrics":["F1","Micro F1"],"metric_direction":{"note":"inferred from the metric name only (the archive records no direction); null = not inferred, chart draws points only","by_metric":{"F1":"higher","Micro F1":"higher"}},"counts":{"rows":5,"rows_with_code":5,"rows_with_paper_page":5,"rows_dated":5,"rows_using_additional_data":0},"rows":[{"rank_in_archive_order":1,"model":"BioLinkBERT (large)","metrics":{"F1":"88.1","Micro F1":"84.87"},"uses_additional_data":false,"paper_date":"2022-03-29","paper":"/paper/linkbert-pretraining-language-models-with","paper_url":"https://arxiv.org/abs/2203.15827v1","paper_title":"LinkBERT: Pretraining Language Models with Document Links","code":"https://github.com/michiyasunaga/LinkBERT","n_code_links":1,"syntology":{"n_ran":0,"n_unverified":14,"n_samples":14,"n_pointer_only_licence":0}},{"rank_in_archive_order":2,"model":"NCBI_BERT(large) (P)","metrics":{"F1":"87.3"},"uses_additional_data":false,"paper_date":"2019-06-13","paper":"/paper/transfer-learning-in-biomedical-natural","paper_url":"https://arxiv.org/abs/1906.05474v2","paper_title":"Transfer Learning in Biomedical Natural Language Processing: An Evaluation of BERT and ELMo on Ten Benchmarking Datasets","code":"https://github.com/ncbi-nlp/NCBI_BERT","n_code_links":4,"syntology":{"n_ran":0,"n_unverified":2,"n_samples":2,"n_pointer_only_licence":0}},{"rank_in_archive_order":3,"model":"SciFive-large","metrics":{"F1":"86.08"},"uses_additional_data":false,"paper_date":"2021-05-28","paper":"/paper/scifive-a-text-to-text-transformer-model-for","paper_url":"https://arxiv.org/abs/2106.03598v1","paper_title":"SciFive: a text-to-text transformer model for biomedical literature","code":"https://github.com/justinphan3110/SciFive","n_code_links":1,"syntology":{"n_ran":0,"n_unverified":5,"n_samples":5,"n_pointer_only_licence":0}},{"rank_in_archive_order":4,"model":"BioGPT","metrics":{"Micro F1":"85.12"},"uses_additional_data":false,"paper_date":"2022-10-19","paper":"/paper/biogpt-generative-pre-trained-transformer-for","paper_url":"https://arxiv.org/abs/2210.10341v3","paper_title":"BioGPT: Generative Pre-trained Transformer for Biomedical Text Generation and Mining","code":"https://github.com/huggingface/transformers","n_code_links":4,"syntology":null},{"rank_in_archive_order":5,"model":"PubMedBERT uncased","metrics":{"Micro F1":"82.32"},"uses_additional_data":false,"paper_date":"2020-07-31","paper":"/paper/domain-specific-language-model-pretraining","paper_url":"https://arxiv.org/abs/2007.15779v6","paper_title":"Domain-Specific Language Model Pretraining for Biomedical Natural Language Processing","code":"https://github.com/bionlu-coling2024/biomed-ner-intent_detection","n_code_links":2,"syntology":null}],"since_archive":{"present":false,"note":"No Syntology-extracted rows are published in this build."},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per row: N of M harvested code samples from that row's paper executed on a synthesized fixture; the other M-N are unverified. Not a reproduction of the row's number; not a correctness claim. n_pointer_only_licence counts samples the site points at rather than redistributes (a licence axis, independent of ran/unverified).","rows_with_graph_line":3,"rows_with_any_sample_ran":0,"distinct_papers_with_graph_line":3,"distinct_papers_with_any_sample_ran":0,"samples_over_distinct_papers":{"n_ran":0,"n_unverified":21,"n_samples":21,"n_pointer_only_licence":0,"note":"each paper (arXiv id) counted once, however many rows it is behind; this is the page-level figure"},"samples_row_weighted":{"n_ran":0,"n_unverified":21,"n_samples":21,"n_pointer_only_licence":0,"note":"row-weighted: a paper behind several rows is counted once per row; inflated relative to samples_over_distinct_papers by design, kept for readers summing the per-row syntology blocks"}}}