{"url":"/dataset/reuters-21578","name":"Reuters-21578","full_name":null,"description_markdown":"The **Reuters-21578** dataset is a collection of documents with news articles. The original corpus has 10,369 documents and a vocabulary of 29,930 words.\r\n\r\nSource: [Topic Model Based Multi-Label Classification from the Crowd](https://arxiv.org/abs/1604.00783)","description_withheld":null,"homepage":"http://kdd.ics.uci.edu/databases/reuters21578/reuters21578.html","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":null,"title":"Reuters-21578","first_author":null,"url":"http://www.daviddlewis.com/resources/testcollections/reuters21578"},"license":{"name":"Custom (research-only, attribution)","url":"http://kdd.ics.uci.edu/databases/reuters21578/README.txt"},"modalities":[{"name":"Graphs","url":"/datasets/modality/graphs"}],"tasks":[{"name":"Text Classification","url":"/task/text-classification","datasets_with_task":"/datasets/task/text-classification"},{"name":"Unsupervised Anomaly Detection","url":"/task/unsupervised-anomaly-detection","datasets_with_task":"/datasets/task/unsupervised-anomaly-detection"},{"name":"Document Classification","url":"/task/document-classification","datasets_with_task":"/datasets/task/document-classification"},{"name":"Multi-Label Text Classification","url":"/task/multi-label-text-classification","datasets_with_task":"/datasets/task/multi-label-text-classification"},{"name":"Text Retrieval","url":"/task/text-retrieval","datasets_with_task":"/datasets/task/text-retrieval"},{"name":"Multi-Modal Document Classification","url":"/task/multi-modal-document-classification","datasets_with_task":"/datasets/task/multi-modal-document-classification"},{"name":"Supervised Text Retrieval","url":"/task/supervised-text-retrieval","datasets_with_task":"/datasets/task/supervised-text-retrieval"}],"languages":[],"variants":["Reuters-21578","reuters21578"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/ucirvine/reuters21578","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/reuters21578","frameworks":["tf","pytorch","jax"]}],"num_papers_in_archive":66,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/document-classification-on-reuters-21578","task":"Document Classification","dataset_variant":"Reuters-21578","rows":8,"metrics":["Accuracy","F1"],"first_row_in_archive_order":{"model":"ApproxRepSet","paper":"/paper/rep-the-set-neural-networks-for-learning-set","metrics":{"Accuracy":"97.17"},"code_links":[{"title":"giannisnik/repset","url":"https://github.com/giannisnik/repset"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/multi-label-text-classification-on-reuters-1","task":"Multi-Label Text Classification","dataset_variant":"Reuters-21578","rows":7,"metrics":["Micro-F1"],"first_row_in_archive_order":{"model":"TIACBM","paper":"/paper/task-informed-anti-curriculum-by-masking","metrics":{"Micro-F1":"91.2±0.20"},"code_links":[{"title":"jarcaandrei/tiacbm","url":"https://github.com/jarcaandrei/tiacbm"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/text-retrieval-on-reuters-21578","task":"Text Retrieval","dataset_variant":"Reuters-21578","rows":3,"metrics":["Precision@100"],"first_row_in_archive_order":{"model":"VDSH","paper":"/paper/variational-deep-semantic-hashing-for-text","metrics":{"Precision@100":"0.7753"},"code_links":[{"title":"unsuthee/VariationalDeepSemanticHashing","url":"https://github.com/unsuthee/VariationalDeepSemanticHashing"},{"title":"J-zin/SNUH","url":"https://github.com/J-zin/SNUH"},{"title":"MindSpore-scientific/code-9","url":"https://github.com/MindSpore-scientific/code-9/tree/main/SNUH"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/unsupervised-anomaly-detection-on-reuters-1","task":"Unsupervised Anomaly Detection","dataset_variant":"Reuters-21578","rows":1,"metrics":["AUC (outlier ratio = 0.5)"],"first_row_in_archive_order":{"model":"RSRAE","paper":"/paper/robust-subspace-recovery-layer-for","metrics":{"AUC (outlier ratio = 0.5)":"0.849"},"code_links":[{"title":"dmzou/RSRAE","url":"https://github.com/dmzou/RSRAE"},{"title":"marrrcin/rsrlayer-pytorch","url":"https://github.com/marrrcin/rsrlayer-pytorch"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/text-classification-on-reuters21578","task":"Text Classification","dataset_variant":"reuters21578","rows":0,"metrics":["Accuracy","F1"],"first_row_in_archive_order":null,"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/task-informed-anti-curriculum-by-masking","title":"Task-Informed Anti-Curriculum by Masking Improves Downstream Performance on Text","date":"2025-02-18","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/co-attention-network-with-label-embedding-for","title":"Co-attention network with label embedding for text classification","date":"2021-11-04","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/balancing-methods-for-multi-label-text","title":"Balancing Methods for Multi-label Text Classification with Long-Tailed Class Distribution","date":"2021-09-10","rows_on_this_dataset":3,"code_links":2,"syntology":null},{"paper":"/paper/text-classification-with-word-embedding","title":"Text classification with word embedding regularization and soft similarity measure","date":"2020-03-10","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/magnet-multi-label-text-classification-using","title":"MAGNET: Multi-Label Text Classification using Attention-based Graph Neural Network","date":"2020-02-24","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/speeding-up-word-movers-distance-and-its","title":"Speeding up Word Mover's Distance and its variants via properties of distances between embeddings","date":"2019-12-01","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/improving-document-classification-with-multi","title":"Improving Document Classification with Multi-Sense Embeddings","date":"2019-11-18","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/a-binary-variational-autoencoder-for-hashing","title":"A Binary Variational Autoencoder for Hashing","date":"2019-10-22","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/rethinking-complex-neural-network","title":"Rethinking Complex Neural Network Architectures for Document Classification","date":"2019-06-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/docbert-bert-for-document-classification","title":"DocBERT: BERT for Document Classification","date":"2019-04-17","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":14,"samples_ran":5,"samples_unverified":9,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/rep-the-set-neural-networks-for-learning-set","title":"Rep the Set: Neural Networks for Learning Set Representations","date":"2019-04-03","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/robust-subspace-recovery-layer-for","title":"Robust Subspace Recovery Layer for Unsupervised Anomaly Detection","date":"2019-03-30","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/vector-of-locally-aggregated-word-embeddings","title":"Vector of Locally-Aggregated Word Embeddings (VLAWE): A Novel Document-level Representation","date":"2019-02-23","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/variational-deep-semantic-hashing-for-text","title":"Variational Deep Semantic Hashing for Text Documents","date":"2017-08-11","rows_on_this_dataset":1,"code_links":3,"syntology":null}],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":2,"samples_harvested":15,"samples_ran":5,"samples_unverified":10,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}