{"url":"/dataset/yahoo-answers","name":"Yahoo! Answers","full_name":null,"description_markdown":"The Yahoo! Answers topic classification dataset is constructed using 10 largest main categories. Each class contains 140,000 training samples and 6,000 testing samples. Therefore, the total number of training samples is 1,400,000 and testing samples 60,000 in this dataset. From all the answers and other meta-information, we only used the best answer content and the main category information.\r\nSource:[github](https://github.com/LC-John/Yahoo-Answers-Topic-Classification-Dataset/tree/master/dataset)","description_withheld":null,"homepage":"https://github.com/LC-John/Yahoo-Answers-Topic-Classification-Dataset","introduced_date":"2015-09-04","introduced_date_note":null,"introduced_by":{"paper":"/paper/character-level-convolutional-networks-for","title":"Character-level Convolutional Networks for Text Classification","first_author":"Xiang Zhang","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Text Classification","url":"/task/text-classification","datasets_with_task":"/datasets/task/text-classification"},{"name":"Unsupervised Text Classification","url":"/task/unsupervised-text-classification","datasets_with_task":"/datasets/task/unsupervised-text-classification"},{"name":"Topic Classification","url":"/task/topic-classification","datasets_with_task":"/datasets/task/topic-classification"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["Yahoo! Answers"],"data_loaders":[],"num_papers_in_archive":132,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/text-classification-on-yahoo-answers","task":"Text Classification","dataset_variant":"Yahoo! Answers","rows":10,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"BERT-ITPT-FiT","paper":"/paper/how-to-fine-tune-bert-for-text-classification","metrics":{"Accuracy":"77.62"},"code_links":[{"title":"xuyige/BERT4doc-Classification","url":"https://github.com/xuyige/BERT4doc-Classification"},{"title":"ongunuzaymacar/comparatively-finetuning-bert","url":"https://github.com/ongunuzaymacar/comparatively-finetuning-bert"},{"title":"uzaymacar/comparatively-finetuning-bert","url":"https://github.com/uzaymacar/comparatively-finetuning-bert"},{"title":"helmy-elrais/RoBERT_Recurrence_over_BERT","url":"https://github.com/helmy-elrais/RoBERT_Recurrence_over_BERT"},{"title":"GeorgeLuImmortal/Hierarchical-BERT-Model-with-Limited-Labelled-Data","url":"https://github.com/GeorgeLuImmortal/Hierarchical-BERT-Model-with-Limited-Labelled-Data"},{"title":"heraclex12/VLSP2020-Fake-News-Detection","url":"https://github.com/heraclex12/VLSP2020-Fake-News-Detection"},{"title":"bcaitech1/p4-dkt-no_caffeine_no_gain","url":"https://github.com/bcaitech1/p4-dkt-no_caffeine_no_gain"},{"title":"soarsmu/BiasFinder","url":"https://github.com/soarsmu/BiasFinder"},{"title":"qinhanmin2014/fine-tune-bert-for-text-classification","url":"https://github.com/qinhanmin2014/fine-tune-bert-for-text-classification"},{"title":"Domminique/Deploy-BERT-for-Sentiment-Analysis-with-FastAPI-","url":"https://github.com/Domminique/Deploy-BERT-for-Sentiment-Analysis-with-FastAPI-"},{"title":"Derposoft/ai-educator","url":"https://github.com/Derposoft/ai-educator"},{"title":"arctic-yen/Google_QUEST_Q-A_Labeling","url":"https://github.com/arctic-yen/Google_QUEST_Q-A_Labeling"},{"title":"sahil00199/KYC","url":"https://github.com/sahil00199/KYC"},{"title":"jyp1111/sentiment_analysis","url":"https://github.com/jyp1111/sentiment_analysis"},{"title":"saproovarun/Google-Quest-Q-A","url":"https://github.com/saproovarun/Google-Quest-Q-A"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/unsupervised-text-classification-on-yahoo","task":"Unsupervised Text Classification","dataset_variant":"Yahoo! Answers","rows":1,"metrics":["F1-score"],"first_row_in_archive_order":{"model":"Lbl2TransformerVec","paper":"/paper/evaluating-unsupervised-text-classification","metrics":{"F1-score":"55.84"},"code_links":[{"title":"sebischair/lbl2vec","url":"https://github.com/sebischair/lbl2vec"},{"title":"sebischair/medical-abstracts-tc-corpus","url":"https://github.com/sebischair/medical-abstracts-tc-corpus"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/evaluating-unsupervised-text-classification","title":"Evaluating Unsupervised Text Classification: Zero-shot and Similarity-based Approaches","date":"2022-11-29","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/sampling-bias-in-deep-active-classification","title":"Sampling Bias in Deep Active Classification: An Empirical Study","date":"2019-09-20","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/delta-a-deep-learning-based-language","title":"DELTA: A DEep learning based Language Technology plAtform","date":"2019-08-02","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/how-to-fine-tune-bert-for-text-classification","title":"How to Fine-Tune BERT for Text Classification?","date":"2019-05-14","rows_on_this_dataset":1,"code_links":15,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":18,"samples_ran":6,"samples_unverified":12,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/learning-to-remember-more-with-less","title":"Learning to Remember More with Less Memorization","date":"2019-01-05","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/explicit-interaction-model-towards-text","title":"Explicit Interaction Model towards Text Classification","date":"2018-11-23","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/compositional-coding-capsule-network-with-k","title":"Compositional Coding Capsule Network with K-Means Routing for Text Classification","date":"2018-10-22","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/disconnected-recurrent-neural-networks-for","title":"Disconnected Recurrent Neural Networks for Text Categorization","date":"2018-07-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/baseline-needs-more-love-on-simple-word","title":"Baseline Needs More Love: On Simple Word-Embedding-Based Models and Associated Pooling Mechanisms","date":"2018-05-24","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/abstractive-text-classification-using","title":"Abstractive Text Classification Using Sequence-to-convolution Neural Networks","date":"2018-05-20","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/bag-of-tricks-for-efficient-text","title":"Bag of Tricks for Efficient Text Classification","date":"2016-07-06","rows_on_this_dataset":1,"code_links":65,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":2,"samples_unverified":7,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":5,"samples_harvested":34,"samples_ran":13,"samples_unverified":21,"pointer_only_for_licence":10,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}