{"url":"/dataset/mtnt","name":"MTNT","full_name":null,"description_markdown":"The Machine Translation of Noisy Text (**MTNT**) dataset is a Machine Translation dataset that consists of noisy comments on Reddit and professionally sourced translation. The translation are between French, Japanese and French, with between 7k and 37k sentence per language pair.\r\n\r\nSource: [https://arxiv.org/abs/1809.00388](https://arxiv.org/abs/1809.00388)\r\nImage Source: [https://github.com/pmichel31415/mtnt](https://github.com/pmichel31415/mtnt)","description_withheld":null,"homepage":"https://github.com/pmichel31415/mtnt","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/mtnt-a-testbed-for-machine-translation-of","title":"MTNT: A Testbed for Machine Translation of Noisy Text","first_author":"Paul Michel","url":null},"license":{"name":"Custom","url":"https://github.com/pmichel31415/mtnt#license"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Domain Adaptation","url":"/task/domain-adaptation","datasets_with_task":"/datasets/task/domain-adaptation"},{"name":"Text Generation","url":"/task/text-generation","datasets_with_task":"/datasets/task/text-generation"},{"name":"Machine Translation","url":"/task/machine-translation","datasets_with_task":"/datasets/task/machine-translation"}],"languages":[{"name":"English","url":"/datasets/language/english"},{"name":"French","url":"/datasets/language/french"},{"name":"Japanese","url":"/datasets/language/japanese"}],"variants":["MTNT"],"data_loaders":[{"repo":"https://github.com/pmichel31415/mtnt","url":"https://github.com/pmichel31415/mtnt","frameworks":[]}],"num_papers_in_archive":52,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}