{"url":"/dataset/ags-corpus","name":"AG’s Corpus","full_name":"AG's corpus of news articlesNews","description_markdown":"Antonio Gulli’s corpus of news articles is a collection of more than 1 million news articles. The articles have been gathered from more than 2000  news sources by ComeToMyHead in more than 1 year of activity. ComeToMyHead is an academic news search engine which has been running since July, 2004.\r\nThe dataset is provided by the academic comunity for research purposes in data mining (clustering, classification, etc), information retrieval (ranking, search, etc), xml, data compression, data streaming, and any other non - commercial activity.\r\n\r\nA subset of this corpus, AG News, consisting of the 4 largest classes is a popular topic classification dataset.\r\n\r\nSource: [AG's corpus of news articles](http://groups.di.unipi.it/~gulli/AG_corpus_of_news_articles.html)","description_withheld":null,"homepage":"http://groups.di.unipi.it/~gulli/AG_corpus_of_news_articles.html","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":null,"title":"AG's corpus of news articles","first_author":null,"url":"http://groups.di.unipi.it/~gulli/AG_corpus_of_news_articles.html"},"license":{"name":"Custom","url":"http://groups.di.unipi.it/~gulli/AG_corpus_of_news_articles.html"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Text Classification","url":"/task/text-classification","datasets_with_task":"/datasets/task/text-classification"},{"name":"Sentiment Analysis","url":"/task/sentiment-analysis","datasets_with_task":"/datasets/task/sentiment-analysis"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["AG’s Corpus"],"data_loaders":[],"num_papers_in_archive":2,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}