{"url":"/dataset/yelp-review-polarity","name":"Yelp Review Polarity","full_name":null,"description_markdown":"The Yelp Reviews Polarity dataset is obtained from the Yelp Dataset Challenge in 2015 (1,569,264 samples that have review text).\r\n\r\nThe polarity label is constructed by considering stars 1 and 2 negative, and 3 and 4 positive.\r\n\r\nThe polarity dataset has 280,000 training samples and 19,000 test samples in each polarity.","description_withheld":null,"homepage":"","introduced_date":"2015-09-04","introduced_date_note":null,"introduced_by":{"paper":"/paper/character-level-convolutional-networks-for","title":"Character-level Convolutional Networks for Text Classification","first_author":"Xiang Zhang","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Text Classification","url":"/task/text-classification","datasets_with_task":"/datasets/task/text-classification"},{"name":"Sentiment Classification","url":"/task/sentiment-classification","datasets_with_task":"/datasets/task/sentiment-classification"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["yelp_polarity","Yelp Review Polarity"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/fancyzhx/yelp_polarity","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/yelp_polarity","frameworks":["tf","pytorch","jax"]}],"num_papers_in_archive":37,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}