{"url":"/dataset/marc","name":"MARC","full_name":"Multilingual Amazon Reviews Corpus","description_markdown":"Multilingual Amazon Reviews Corpus (MARC) is a large-scale collection of Amazon reviews for multilingual text classification. The corpus contains reviews in English, Japanese, German, French, Spanish, and Chinese, which were collected between 2015 and 2019. Each record in the dataset contains the review text, the review title, the star rating, an anonymized reviewer ID, an anonymized product ID, and the coarse-grained product category (e.g., 'books', 'appliances', etc.) The corpus is balanced across the 5 possible star ratings, so each rating constitutes 20% of the reviews in each language. For each language, there are 200,000, 5,000, and 5,000 reviews in the training, development, and test sets, respectively.","description_withheld":null,"homepage":"https://huggingface.co/datasets/mteb/amazon_reviews_multi","introduced_date":"2020-10-06","introduced_date_note":null,"introduced_by":{"paper":"/paper/the-multilingual-amazon-reviews-corpus","title":"The Multilingual Amazon Reviews Corpus","first_author":"Phillip Keung","url":null},"license":null,"modalities":[],"tasks":[],"languages":[],"variants":["MARC"],"data_loaders":[],"num_papers_in_archive":35,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}