{"url":"/dataset/the-tatoeba-translation-challenge","name":"TTC","full_name":"Tatoeba Translation Challenge","description_markdown":"This is a challenge set for machine translation that contains 32G translation units in 2,539 bitexts. The whole data set covers 487 languages linked to each other in 4,024 language pairs. The package includes a release of 657 test sets derived from Tatoeba.org that cover 138 languages. Training data is compiled from various sources collected within the OPUS project.","description_withheld":null,"homepage":"https://github.com/Helsinki-NLP/Tatoeba-Challenge/","introduced_date":"2020-10-13","introduced_date_note":null,"introduced_by":{"paper":"/paper/the-tatoeba-translation-challenge-realistic","title":"The Tatoeba Translation Challenge -- Realistic Data Sets for Low Resource and Multilingual MT","first_author":"Jörg Tiedemann","url":null},"license":null,"modalities":[],"tasks":[],"languages":[],"variants":["TTC"],"data_loaders":[],"num_papers_in_archive":2,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}