{"url":"/dataset/mage","name":"MAGE","full_name":null,"description_markdown":"The MAGE dataset provides a large set of generated texts using 27 LLMs from seven different groups: OpenAI GPT, LLaMA, GLM130B, FLAN-T5, OPT, BigScience, and EleutherAI. In total, the dataset contains 432,682 texts, along with two additional sets. The first is an additional test set with texts from unseen domains generated by an unseen model, namely GPT-4. The second set is designed to evaluate the robustness of detectors against paraphrasing attacks. To achieve this, the GPT-3.5-turbo model was employed to paraphrase the sentences from the first set, with all paraphrased texts treated as machine-generated.","description_withheld":null,"homepage":"https://github.com/yafuly/MAGE","introduced_date":"2023-05-22","introduced_date_note":null,"introduced_by":{"paper":"/paper/deepfake-text-detection-in-the-wild","title":"MAGE: Machine-generated Text Detection in the Wild","first_author":"Yafu Li","url":null},"license":{"name":"MIT","url":"https://opensource.org/license/mit"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Text Classification","url":"/task/text-classification","datasets_with_task":"/datasets/task/text-classification"},{"name":"Binary text classification","url":"/task/binary-text-classification","datasets_with_task":"/datasets/task/binary-text-classification"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["MAGE","MAGE (Arbitrary-domains & Arbitrary-models)"],"data_loaders":[],"num_papers_in_archive":11,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/binary-text-classification-on-mage-arbitrary","task":"Binary text classification","dataset_variant":"MAGE (Arbitrary-domains & Arbitrary-models)","rows":2,"metrics":["Average Recall"],"first_row_in_archive_order":{"model":"GigaCheck (Mistral-7B)","paper":"/paper/gigacheck-detecting-llm-generated-content","metrics":{"Average Recall":"0.9611"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/gigacheck-detecting-llm-generated-content","title":"GigaCheck: Detecting LLM-generated Content","date":"2024-10-31","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/deepfake-text-detection-in-the-wild","title":"MAGE: Machine-generated Text Detection in the Wild","date":"2023-05-22","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}