{"url":"/dataset/wmt-2015","name":"WMT 2015","full_name":null,"description_markdown":"**WMT 2015** is a collection of datasets used in shared tasks of the Tenth Workshop on Statistical Machine Translation. The workshop featured five tasks:\r\n\r\n* a news translation task,\r\n* a metrics task,\r\n* a tuning task,\r\n* a quality estimation task,\r\n* an automatic post-editing task.\r\n\r\nSource: [https://www.aclweb.org/anthology/W15-3001.pdf](https://www.aclweb.org/anthology/W15-3001.pdf)","description_withheld":null,"homepage":"http://www.statmt.org/wmt15/index.html","introduced_date":"2015-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/findings-of-the-2015-workshop-on-statistical","title":"Findings of the 2015 Workshop on Statistical Machine Translation","first_author":"Ond{\\v{r}}ej Bojar","url":null},"license":{"name":"Unknown","url":null},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Machine Translation","url":"/task/machine-translation","datasets_with_task":"/datasets/task/machine-translation"},{"name":"Translation deu-eng","url":"/task/translation-deu-eng","datasets_with_task":"/datasets/task/translation-deu-eng"},{"name":"Translation eng-deu","url":"/task/translation-eng-deu","datasets_with_task":"/datasets/task/translation-eng-deu"}],"languages":[],"variants":["WMT2015 English-German","WMT2015 English-Russian","WMT 2015","newstest2015-deen","newstest2015-ende","newstest2015-deen deu-eng","newstest2015-ende eng-deu"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/wmt/wmt15","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/wmt15","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/tensorflow/datasets","url":"https://www.tensorflow.org/datasets/catalog/wmt15_translate","frameworks":["tf","jax"]}],"num_papers_in_archive":33,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/machine-translation-on-wmt2015-english-german","task":"Machine Translation","dataset_variant":"WMT2015 English-German","rows":6,"metrics":["BLEU score"],"first_row_in_archive_order":{"model":"ByteNet","paper":"/paper/neural-machine-translation-in-linear-time","metrics":{"BLEU score":"26.3"},"code_links":[{"title":"paarthneekhara/byteNet-tensorflow","url":"https://github.com/paarthneekhara/byteNet-tensorflow"},{"title":"microsoft/protein-sequence-models","url":"https://github.com/microsoft/protein-sequence-models"},{"title":"randomrandom/deep-atrous-cnn-sentiment","url":"https://github.com/randomrandom/deep-atrous-cnn-sentiment"},{"title":"kingstarcraft/speech-to-text-wavenet2","url":"https://github.com/kingstarcraft/speech-to-text-wavenet2"},{"title":"sriharireddypusapati/speech-to-text-wavenet2","url":"https://github.com/sriharireddypusapati/speech-to-text-wavenet2"},{"title":"kinimod23/ATS_Project","url":"https://github.com/kinimod23/ATS_Project"},{"title":"Vikas-Sony/speech-to-text","url":"https://github.com/Vikas-Sony/speech-to-text"},{"title":"freedombenLiu/speech-to-text-wavenet","url":"https://github.com/freedombenLiu/speech-to-text-wavenet"},{"title":"adityaagrawal7/speech-to-text-wavenet","url":"https://github.com/adityaagrawal7/speech-to-text-wavenet"},{"title":"liguigui/speech-to-text-wavenet","url":"https://github.com/liguigui/speech-to-text-wavenet"},{"title":"Shivendra-psc/speechbot","url":"https://github.com/Shivendra-psc/speechbot"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/machine-translation-on-wmt2015-english","task":"Machine Translation","dataset_variant":"WMT2015 English-Russian","rows":1,"metrics":["BLEU score"],"first_row_in_archive_order":{"model":"C2-50k Segmentation","paper":"/paper/neural-machine-translation-of-rare-words-with","metrics":{"BLEU score":"20.9"},"code_links":[{"title":"facebookresearch/fairseq","url":"https://github.com/facebookresearch/fairseq"},{"title":"google/sentencepiece","url":"https://github.com/google/sentencepiece"},{"title":"karpathy/minbpe","url":"https://github.com/karpathy/minbpe"},{"title":"rsennrich/subword-nmt","url":"https://github.com/rsennrich/subword-nmt"},{"title":"VKCOM/YouTokenToMe","url":"https://github.com/VKCOM/YouTokenToMe"},{"title":"glample/fastBPE","url":"https://github.com/glample/fastBPE"},{"title":"salesforce/GeDi","url":"https://github.com/salesforce/GeDi"},{"title":"EdinburghNLP/code-docstring-corpus","url":"https://github.com/EdinburghNLP/code-docstring-corpus"},{"title":"Avmb/code-docstring-corpus","url":"https://github.com/Avmb/code-docstring-corpus"},{"title":"nyu-dl/dl4mt-cdec","url":"https://github.com/nyu-dl/dl4mt-cdec"},{"title":"nyu-dl/dl4mt-c2c","url":"https://github.com/nyu-dl/dl4mt-c2c"},{"title":"ThAIKeras/bert","url":"https://github.com/ThAIKeras/bert"},{"title":"nyu-dl/dl4mt-simul-trans","url":"https://github.com/nyu-dl/dl4mt-simul-trans"},{"title":"Automattic/wp-translate","url":"https://github.com/Automattic/wp-translate"},{"title":"SeonbeomKim/Python-Bype_Pair_Encoding","url":"https://github.com/SeonbeomKim/Python-Bype_Pair_Encoding"},{"title":"thinkwee/DPP_CNN_Summarization","url":"https://github.com/thinkwee/DPP_CNN_Summarization"},{"title":"SeonbeomKim/Python-Byte_Pair_Encoding","url":"https://github.com/SeonbeomKim/Python-Byte_Pair_Encoding"},{"title":"PaulSudarshan/Language-Classification-Using-Naive-Bayes-Algorithm","url":"https://github.com/PaulSudarshan/Language-Classification-Using-Naive-Bayes-Algorithm"},{"title":"johnr0/TaleBrush-backend","url":"https://github.com/johnr0/TaleBrush-backend"},{"title":"kh-mo/QA_wikisql","url":"https://github.com/kh-mo/QA_wikisql"},{"title":"simonjisu/NMT","url":"https://github.com/simonjisu/NMT"},{"title":"siyuofzhou/CNNSeqToSeq","url":"https://github.com/siyuofzhou/CNNSeqToSeq"},{"title":"lkfo415579/MT-Readling-List","url":"https://github.com/lkfo415579/MT-Readling-List"},{"title":"HarshKhandelwal1552/Language_classifier_with_Naive_Bayes","url":"https://github.com/HarshKhandelwal1552/Language_classifier_with_Naive_Bayes"},{"title":"Xinsen-Zhang/transformer","url":"https://github.com/Xinsen-Zhang/transformer"},{"title":"ksulima/Unsupervised-method-to-NPL-Polish-language","url":"https://github.com/ksulima/Unsupervised-method-to-NPL-Polish-language"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/unsupervised-neural-machine-translation","title":"Unsupervised Neural Machine Translation","date":"2017-10-30","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":6,"samples_unverified":0,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/neural-machine-translation-in-linear-time","title":"Neural Machine Translation in Linear Time","date":"2016-10-31","rows_on_this_dataset":1,"code_links":11,"syntology":null},{"paper":"/paper/a-character-level-decoder-without-explicit","title":"A Character-Level Decoder without Explicit Segmentation for Neural Machine Translation","date":"2016-03-19","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":1,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/neural-machine-translation-of-rare-words-with","title":"Neural Machine Translation of Rare Words with Subword Units","date":"2015-08-31","rows_on_this_dataset":2,"code_links":26,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":30,"samples_ran":21,"samples_unverified":9,"pointer_only_for_licence":21,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":3,"samples_harvested":39,"samples_ran":28,"samples_unverified":11,"pointer_only_for_licence":27,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}