{"url":"/dataset/rcv1","name":"RCV1","full_name":"Reuters Corpus Volume 1","description_markdown":"The **RCV1** dataset is a benchmark dataset on text categorization. It is a collection of newswire articles producd by Reuters in 1996-1997. It contains 804,414 manually labeled newswire documents, and categorized with respect to three controlled vocabularies: industries, topics and regions.\r\n\r\nSource: [Random Projections for Linear Support Vector Machines](https://arxiv.org/abs/1211.6085)\r\nImage Source: [https://www.nasdaq.com/publishers/reuters](https://www.nasdaq.com/publishers/reuters)","description_withheld":null,"homepage":"http://www.ai.mit.edu/projects/jmlr/papers/volume5/lewis04a/lyrl2004_rcv1v2_README.htm","introduced_date":"2004-01-01","introduced_date_note":null,"introduced_by":{"paper":null,"title":"RCV1: A New Benchmark Collection for Text Categorization Research","first_author":null,"url":"http://jmlr.org/papers/volume5/lewis04a/lewis04a.pdf"},"license":{"name":"Custom","url":"http://www.ai.mit.edu/projects/jmlr/papers/volume5/lewis04a/lyrl2004_rcv1v2_README.htm"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Text Classification","url":"/task/text-classification","datasets_with_task":"/datasets/task/text-classification"},{"name":"Multi-Label Text Classification","url":"/task/multi-label-text-classification","datasets_with_task":"/datasets/task/multi-label-text-classification"},{"name":"Hierarchical Multi-label Classification","url":"/task/hierarchical-multi-label-classification","datasets_with_task":"/datasets/task/hierarchical-multi-label-classification"},{"name":"Cross-Lingual Document Classification","url":"/task/cross-lingual-document-classification","datasets_with_task":"/datasets/task/cross-lingual-document-classification"}],"languages":[],"variants":["RCV1-v2","Reuters RCV1/RCV2 English-to-German","Reuters RCV1/RCV2 German-to-English","RCV1"],"data_loaders":[{"repo":"https://github.com/wangzemin63/dataset","url":"https://github.com/wangzemin63/dataset","frameworks":["pytorch"]}],"num_papers_in_archive":336,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/hierarchical-multi-label-classification-on-17","task":"Hierarchical Multi-label Classification","dataset_variant":"RCV1-v2","rows":7,"metrics":["Macro F1","Micro F1"],"first_row_in_archive_order":{"model":"HiDEC+HBM Loss","paper":"/paper/hierarchy-aware-biased-bound-margin-loss","metrics":{"Macro F1":"71.47±0.20","Micro F1":"87.81±0.09"},"code_links":[{"title":"whitepurple/HBM-loss-for-HTC","url":"https://github.com/whitepurple/HBM-loss-for-HTC"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/text-classification-on-rcv1","task":"Text Classification","dataset_variant":"RCV1","rows":4,"metrics":["Accuracy","Macro F1","Micro F1","P@1","P@3","P@5","nDCG@1","nDCG@3","nDCG@5"],"first_row_in_archive_order":{"model":"oh-CNN + two LSTM tv-embed.","paper":"/paper/supervised-and-semi-supervised-text","metrics":{"Accuracy":"92.85"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/cross-lingual-document-classification-on-12","task":"Cross-Lingual Document Classification","dataset_variant":"Reuters RCV1/RCV2 English-to-German","rows":3,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Biinclusion (Euro500kReuters)","paper":"/paper/leveraging-monolingual-data-for-crosslingual","metrics":{"Accuracy":"92.7"},"code_links":[{"title":"ogh/binclusion","url":"https://github.com/ogh/binclusion"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/cross-lingual-document-classification-on-13","task":"Cross-Lingual Document Classification","dataset_variant":"Reuters RCV1/RCV2 German-to-English","rows":3,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Biinclusion (Euro500kReuters)","paper":"/paper/leveraging-monolingual-data-for-crosslingual","metrics":{"Accuracy":"84.4"},"code_links":[{"title":"ogh/binclusion","url":"https://github.com/ogh/binclusion"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/multi-label-text-classification-on-rcv1","task":"Multi-Label Text Classification","dataset_variant":"RCV1","rows":1,"metrics":["Macro-F1","Micro-F1"],"first_row_in_archive_order":{"model":"HiddeN","paper":"/paper/joint-learning-of-hyperbolic-label-embeddings","metrics":{"Macro-F1":"47.3","Micro-F1":"79.3"},"code_links":[{"title":"soumyac1999/hyperbolic-label-emb-for-hmc","url":"https://github.com/soumyac1999/hyperbolic-label-emb-for-hmc"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/multi-label-text-classification-on-rcv1-v2-1","task":"Multi-Label Text Classification","dataset_variant":"RCV1-v2","rows":1,"metrics":["Micro-F1"],"first_row_in_archive_order":{"model":"MAGNET","paper":"/paper/magnet-multi-label-text-classification-using","metrics":{"Micro-F1":"88.5"},"code_links":[{"title":"adrinta/MAGNET","url":"https://github.com/adrinta/MAGNET"},{"title":"akash18tripathi/MAGNET-Multi-Label-Text-Classi-cation-using-Attention-based-Graph-Neural-Network","url":"https://github.com/akash18tripathi/MAGNET-Multi-Label-Text-Classi-cation-using-Attention-based-Graph-Neural-Network"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/hierarchy-aware-biased-bound-margin-loss","title":"Hierarchy-aware Biased Bound Margin Loss Function for Hierarchical Text Classification","date":"2024-08-13","rows_on_this_dataset":4,"code_links":1,"syntology":null},{"paper":"/paper/hill-hierarchy-aware-information-lossless","title":"HILL: Hierarchy-aware Information Lossless Contrastive Learning for Hierarchical Text Classification","date":"2024-03-26","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/hitin-hierarchy-aware-tree-isomorphism","title":"HiTIN: Hierarchy-aware Tree Isomorphism Network for Hierarchical Text Classification","date":"2023-05-24","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":17,"samples_ran":13,"samples_unverified":4,"pointer_only_for_licence":17,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/joint-learning-of-hyperbolic-label-embeddings","title":"Joint Learning of Hyperbolic Label Embeddings for Hierarchical Multi-label Classification","date":"2021-01-13","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/magnet-multi-label-text-classification-using","title":"MAGNET: Multi-Label Text Classification using Attention-based Graph Neural Network","date":"2020-02-24","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/hierarchical-text-classification-with","title":"Hierarchical Text Classification with Reinforced Label Assignment","date":"2019-08-27","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/towards-scalable-and-reliable-capsule","title":"Towards Scalable and Reliable Capsule Networks for Challenging NLP Applications","date":"2019-06-06","rows_on_this_dataset":1,"code_links":5,"syntology":null},{"paper":"/paper/supervised-and-semi-supervised-text","title":"Supervised and Semi-Supervised Text Categorization using LSTM for Region Embeddings","date":"2016-02-07","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/leveraging-monolingual-data-for-crosslingual","title":"Leveraging Monolingual Data for Crosslingual Compositional Word Representations","date":"2014-12-19","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/multilingual-models-for-compositional","title":"Multilingual Models for Compositional Distributed Semantics","date":"2014-04-17","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/multilingual-distributed-representations","title":"Multilingual Distributed Representations without Word Alignment","date":"2013-12-20","rows_on_this_dataset":2,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":2,"samples_harvested":20,"samples_ran":16,"samples_unverified":4,"pointer_only_for_licence":17,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}