{"url":"/dataset/ohsumed","name":"Ohsumed","full_name":null,"description_markdown":"**Ohsumed** includes medical abstracts from the MeSH categories of the year 1991. In [Joachims, 1997] were used the first 20,000 documents divided in 10,000 for training and 10,000 for testing. The specific task was to categorize the 23 cardiovascular diseases categories. After selecting the such category subset, the unique abstract number becomes 13,929 (6,286 for training and 7,643 for testing). As current computers can easily manage larger number of documents we make available all 34,389 cardiovascular diseases abstracts out of 50,216 medical abstracts contained in the year 1991.\r\n\r\nSource: [http://disi.unitn.it/moschitti/corpora.htm](http://disi.unitn.it/moschitti/corpora.htm)","description_withheld":null,"homepage":"http://disi.unitn.it/moschitti/corpora.htm","introduced_date":null,"introduced_date_note":null,"introduced_by":null,"license":{"name":"Unknown","url":null},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Text Classification","url":"/task/text-classification","datasets_with_task":"/datasets/task/text-classification"},{"name":"Information Retrieval","url":"/task/information-retrieval","datasets_with_task":"/datasets/task/information-retrieval"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["Ohsumed"],"data_loaders":[{"repo":"https://github.com/manitprats/NLP","url":"https://github.com/manitprats/NLP","frameworks":["tf"]}],"num_papers_in_archive":11,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/text-classification-on-ohsumed","task":"Text Classification","dataset_variant":"Ohsumed","rows":10,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"RoBERTaGCN","paper":"/paper/bertgcn-transductive-text-classification-by","metrics":{"Accuracy":"72.8"},"code_links":[{"title":"ZeroRin/BertGCN","url":"https://github.com/ZeroRin/BertGCN"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/information-retrieval-on-ohsumed","task":"Information Retrieval","dataset_variant":"Ohsumed","rows":1,"metrics":["NDCG"],"first_row_in_archive_order":{"model":"BERT+CONCEPT FILTER","paper":"/paper/semantic-enrichment-of-pretrained-embedding","metrics":{"NDCG":"0.25"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/bertgcn-transductive-text-classification-by","title":"BertGCN: Transductive Text Classification by Combining GCN and BERT","date":"2021-05-12","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/semantic-enrichment-of-pretrained-embedding","title":"Semantic Enrichment of Pretrained Embedding Output for Unsupervised IR","date":"2021-03-24","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/simple-spectral-graph-convolution","title":"Simple Spectral Graph Convolution","date":"2021-01-01","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/speeding-up-word-movers-distance-and-its","title":"Speeding up Word Mover's Distance and its variants via properties of distances between embeddings","date":"2019-12-01","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/text-level-graph-neural-network-for-text","title":"Text Level Graph Neural Network for Text Classification","date":"2019-10-06","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/graph-star-net-for-generalized-multi-task-1","title":"Graph Star Net for Generalized Multi-Task Learning","date":"2019-06-21","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":1,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/rep-the-set-neural-networks-for-learning-set","title":"Rep the Set: Neural Networks for Learning Set Representations","date":"2019-04-03","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/simplifying-graph-convolutional-networks","title":"Simplifying Graph Convolutional Networks","date":"2019-02-19","rows_on_this_dataset":2,"code_links":7,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":3,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/graph-convolutional-networks-for-text","title":"Graph Convolutional Networks for Text Classification","date":"2018-09-15","rows_on_this_dataset":1,"code_links":9,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":16,"samples_ran":6,"samples_unverified":10,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/on-the-role-of-text-preprocessing-in-neural","title":"On the Role of Text Preprocessing in Neural Network Architectures: An Evaluation Study on Text Categorization and Sentiment Analysis","date":"2017-07-06","rows_on_this_dataset":1,"code_links":3,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":5,"samples_harvested":33,"samples_ran":11,"samples_unverified":22,"pointer_only_for_licence":8,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}