{"url":"/dataset/mgtacademic","name":"MGTAcademic","full_name":null,"description_markdown":"This repository provides a cleaned dataset, which is intended to be used for text classification, language modeling, and AI-generated content detection tasks. The dataset covers various fields such as STEM, Social Sciences, and Humanities, and contains datasets from different categories, each of which has been processed and cleaned for easy use. Move to our codebase fro more information (github)","description_withheld":null,"homepage":"https://huggingface.co/datasets/AITextDetect/AI_Polish_clean","introduced_date":"2024-12-23","introduced_date_note":null,"introduced_by":{"paper":"/paper/on-the-generalization-ability-of-machine","title":"On the Generalization Ability of Machine-Generated Text Detectors","first_author":"Yule Liu","url":null},"license":{"name":"mit","url":null},"modalities":[],"tasks":[{"name":"Classification","url":"/task/classification-1","datasets_with_task":"/datasets/task/classification-1"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["MGTAcademic"],"data_loaders":[],"num_papers_in_archive":2,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}