{"url":"/dataset/diabla","name":"DiaBLa","full_name":null,"description_markdown":"A new English-French test set for the evaluation of Machine Translation (MT) for informal, written bilingual dialogue. The test set contains 144 spontaneous dialogues (5,700+ sentences) between native English and French speakers, mediated by one of two neural MT systems in a range of role-play settings. The dialogues are accompanied by fine-grained sentence-level judgments of MT quality, produced by the dialogue participants themselves, as well as by manually normalised versions and reference translations produced a posteriori. \r\n\r\nSource: [DiaBLa: A Corpus of Bilingual Spontaneous Written Dialogues for Machine Translation](/paper/diabla-a-corpus-of-bilingual-spontaneous)","description_withheld":null,"homepage":"https://github.com/rbawden/DiaBLa-dataset","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/diabla-a-corpus-of-bilingual-spontaneous","title":"DiaBLa: A Corpus of Bilingual Spontaneous Written Dialogues for Machine Translation","first_author":"Rachel Bawden","url":null},"license":null,"modalities":[],"tasks":[{"name":"Text Generation","url":"/task/text-generation","datasets_with_task":"/datasets/task/text-generation"},{"name":"Machine Translation","url":"/task/machine-translation","datasets_with_task":"/datasets/task/machine-translation"}],"languages":[{"name":"English","url":"/datasets/language/english"},{"name":"French","url":"/datasets/language/french"}],"variants":["DiaBLa"],"data_loaders":[{"repo":"https://github.com/rbawden/DiaBLa-dataset","url":"https://github.com/rbawden/DiaBLa-dataset","frameworks":[]}],"num_papers_in_archive":4,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/text-generation-on-diabla","task":"Text Generation","dataset_variant":"DiaBLa","rows":0,"metrics":["acc"],"first_row_in_archive_order":null,"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}