{"url":"/dataset/wmt-2014","name":"WMT 2014","full_name":null,"description_markdown":"**WMT 2014** is a collection of datasets used in shared tasks of the Ninth Workshop on Statistical Machine Translation. The workshop featured four tasks:\r\n\r\n* a news translation task,\r\n* a quality estimation task,\r\n* a metrics task,\r\n* a medical text translation task.\r\n\r\nSource: [https://www.aclweb.org/anthology/W14-3302.pdf](https://www.aclweb.org/anthology/W14-3302.pdf)","description_withheld":null,"homepage":"http://www.statmt.org/wmt14/index.html","introduced_date":"2014-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/findings-of-the-2014-workshop-on-statistical","title":"Findings of the 2014 Workshop on Statistical Machine Translation","first_author":"Ondrej Bojar","url":null},"license":{"name":"Unknown","url":null},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Machine Translation","url":"/task/machine-translation","datasets_with_task":"/datasets/task/machine-translation"},{"name":"Translation deu-eng","url":"/task/translation-deu-eng","datasets_with_task":"/datasets/task/translation-deu-eng"},{"name":"Translation eng-deu","url":"/task/translation-eng-deu","datasets_with_task":"/datasets/task/translation-eng-deu"},{"name":"Unsupervised Machine Translation","url":"/task/unsupervised-machine-translation","datasets_with_task":"/datasets/task/unsupervised-machine-translation"}],"languages":[{"name":"Tamil","url":"/datasets/language/tamil"}],"variants":["WMT-newstest2014-deen","WMT 2014 News","newstest2014-deen eng-deu","newstest2014-deen deu-eng","newstest2014-deen","wmt14","WMT 2014","WMT2014 German-English","WMT2014 English-Czech","WMT 2014 EN-FR","WMT 2014 EN-DE","WMT2014 English-German","WMT2014 French-English","WMT2014 English-French"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/wmt/wmt14","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/wmt14","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/facebookresearch/ParlAI","url":"https://parl.ai/docs/tasks.html#wmt","frameworks":["pytorch"]},{"repo":"https://github.com/tensorflow/datasets","url":"https://www.tensorflow.org/datasets/catalog/wmt14_translate","frameworks":["tf","jax"]}],"num_papers_in_archive":288,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/machine-translation-on-wmt2014-english-german","task":"Machine Translation","dataset_variant":"WMT2014 English-German","rows":91,"metrics":["BLEU score","SacreBLEU","Number of Params","Hardware Burden","Operations per network pass"],"first_row_in_archive_order":{"model":"Transformer Cycle (Rev)","paper":"/paper/lessons-on-parameter-sharing-across-layers-in","metrics":{"BLEU score":"35.14","SacreBLEU":"33.54"},"code_links":[{"title":"takase/share_layer_params","url":"https://github.com/takase/share_layer_params"},{"title":"jaketae/param-share-transformer","url":"https://github.com/jaketae/param-share-transformer"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/machine-translation-on-wmt2014-english-french","task":"Machine Translation","dataset_variant":"WMT2014 English-French","rows":57,"metrics":["BLEU score","SacreBLEU","Hardware Burden","Operations per network pass"],"first_row_in_archive_order":{"model":"Transformer+BT (ADMIN init)","paper":"/paper/very-deep-transformers-for-neural-machine","metrics":{"BLEU score":"46.4","SacreBLEU":"44.4"},"code_links":[{"title":"LiyuanLucasLiu/Transforemr-Clinic","url":"https://github.com/LiyuanLucasLiu/Transforemr-Clinic"},{"title":"LiyuanLucasLiu/Transformer-Clinic","url":"https://github.com/LiyuanLucasLiu/Transformer-Clinic"},{"title":"microsoft/deepnmt","url":"https://github.com/microsoft/deepnmt"},{"title":"namisan/exdeep-nmt","url":"https://github.com/namisan/exdeep-nmt"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/machine-translation-on-wmt2014-german-english","task":"Machine Translation","dataset_variant":"WMT2014 German-English","rows":16,"metrics":["BLEU score"],"first_row_in_archive_order":{"model":"Bi-SimCut","paper":"/paper/bi-simcut-a-simple-strategy-for-boosting-1","metrics":{"BLEU score":"35.15"},"code_links":[{"title":"gpengzhi/Bi-SimCut","url":"https://github.com/gpengzhi/Bi-SimCut"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/unsupervised-machine-translation-on-wmt2014-1","task":"Unsupervised Machine Translation","dataset_variant":"WMT2014 French-English","rows":7,"metrics":["BLEU"],"first_row_in_archive_order":{"model":"GPT-3 175B (Few-Shot)","paper":"/paper/language-models-are-few-shot-learners","metrics":{"BLEU":"39.2"},"code_links":[{"title":"ggml-org/llama.cpp","url":"https://github.com/ggml-org/llama.cpp"},{"title":"ggerganov/llama.cpp","url":"https://github.com/ggerganov/llama.cpp"},{"title":"karpathy/llm.c","url":"https://github.com/karpathy/llm.c"},{"title":"openai/gpt-3","url":"https://github.com/openai/gpt-3"},{"title":"PaddlePaddle/PaddleNLP","url":"https://github.com/PaddlePaddle/PaddleNLP/tree/develop/examples/language_model/gpt-3"},{"title":"EleutherAI/lm_evaluation_harness","url":"https://github.com/EleutherAI/lm_evaluation_harness"},{"title":"EleutherAI/lm-evaluation-harness","url":"https://github.com/EleutherAI/lm-evaluation-harness"},{"title":"EleutherAI/gpt-neo","url":"https://github.com/EleutherAI/gpt-neo"},{"title":"karpathy/build-nanogpt","url":"https://github.com/karpathy/build-nanogpt"},{"title":"ncoop57/gpt-code-clippy","url":"https://github.com/ncoop57/gpt-code-clippy"},{"title":"codedotal/gpt-code-clippy","url":"https://github.com/codedotal/gpt-code-clippy"},{"title":"bigscience-workshop/promptsource","url":"https://github.com/bigscience-workshop/promptsource"},{"title":"shreyashankar/gpt3-sandbox","url":"https://github.com/shreyashankar/gpt3-sandbox"},{"title":"bigscience-workshop/Megatron-DeepSpeed","url":"https://github.com/bigscience-workshop/Megatron-DeepSpeed"},{"title":"NVIDIA/NeMo-Curator","url":"https://github.com/NVIDIA/NeMo-Curator"},{"title":"RUCAIBox/LLMBox","url":"https://github.com/RUCAIBox/LLMBox"},{"title":"hazyresearch/ama_prompting","url":"https://github.com/hazyresearch/ama_prompting"},{"title":"allenai/macaw","url":"https://github.com/allenai/macaw"},{"title":"facebookresearch/anli","url":"https://github.com/facebookresearch/anli"},{"title":"tonyzhaozh/few-shot-learning","url":"https://github.com/tonyzhaozh/few-shot-learning"},{"title":"haiyang-w/git","url":"https://github.com/haiyang-w/git"},{"title":"volcengine/vegiantmodel","url":"https://github.com/volcengine/vegiantmodel"},{"title":"mindspore-ai/models","url":"https://github.com/mindspore-ai/models/tree/master/official/nlp/gpt"},{"title":"asahi417/lmppl","url":"https://github.com/asahi417/lmppl"},{"title":"ethanjperez/true_few_shot","url":"https://github.com/ethanjperez/true_few_shot"},{"title":"ai21labs/lm-evaluation","url":"https://github.com/ai21labs/lm-evaluation"},{"title":"lambert-x/prolab","url":"https://github.com/lambert-x/prolab"},{"title":"asahi417/relbert","url":"https://github.com/asahi417/relbert"},{"title":"grantslatton/llama.cpp","url":"https://github.com/grantslatton/llama.cpp"},{"title":"milmor/GPT","url":"https://github.com/milmor/GPT"},{"title":"um-arm-lab/efficient-eng-2-ltl","url":"https://github.com/um-arm-lab/efficient-eng-2-ltl"},{"title":"abhaskumarsinha/MinimalGPT","url":"https://github.com/abhaskumarsinha/MinimalGPT"},{"title":"turkunlp/megatron-deepspeed","url":"https://github.com/turkunlp/megatron-deepspeed"},{"title":"contextlab/abstract2paper","url":"https://github.com/contextlab/abstract2paper"},{"title":"kyegomez/GPT3","url":"https://github.com/kyegomez/GPT3"},{"title":"Samyu0304/thought-propagation","url":"https://github.com/Samyu0304/thought-propagation"},{"title":"smarton-empower/smarton-ai","url":"https://github.com/smarton-empower/smarton-ai"},{"title":"smile-data/smile","url":"https://github.com/smile-data/smile"},{"title":"postech-ami/smile-dataset","url":"https://github.com/postech-ami/smile-dataset"},{"title":"opengptx/lm-evaluation-harness","url":"https://github.com/opengptx/lm-evaluation-harness"},{"title":"gmum/dl-mo-2021","url":"https://github.com/gmum/dl-mo-2021"},{"title":"fywalter/label-bias","url":"https://github.com/fywalter/label-bias"},{"title":"nlx-group/overlapy","url":"https://github.com/nlx-group/overlapy"},{"title":"openbiolink/promptsource","url":"https://github.com/openbiolink/promptsource"},{"title":"insait-institute/lm-evaluation-harness-bg","url":"https://github.com/insait-institute/lm-evaluation-harness-bg"},{"title":"x-lance/neusym-rag","url":"https://github.com/x-lance/neusym-rag"},{"title":"crazydigger/Callibration-of-GPT","url":"https://github.com/crazydigger/Callibration-of-GPT"},{"title":"abhaskumarsinha/Corpus2GPT","url":"https://github.com/abhaskumarsinha/Corpus2GPT"},{"title":"vilm-ai/viet-llm-eval","url":"https://github.com/vilm-ai/viet-llm-eval"},{"title":"roberttwomey/machine-imagination-workshop","url":"https://github.com/roberttwomey/machine-imagination-workshop"},{"title":"VachanVY/gpt.jax","url":"https://github.com/VachanVY/gpt.jax"},{"title":"neuralmagic/lm-evaluation-harness","url":"https://github.com/neuralmagic/lm-evaluation-harness"},{"title":"scrayish/ML_NLP","url":"https://github.com/scrayish/ML_NLP"},{"title":"sambanova/lm-evaluation-harness","url":"https://github.com/sambanova/lm-evaluation-harness"},{"title":"roberttwomey/machine-imagination-isea","url":"https://github.com/roberttwomey/machine-imagination-isea"},{"title":"ltruncel/Microsoft_Azure_50daysofudacity","url":"https://github.com/ltruncel/Microsoft_Azure_50daysofudacity"},{"title":"ramanakshay/nanogpt","url":"https://github.com/ramanakshay/nanogpt"},{"title":"juletx/lm-evaluation-harness","url":"https://github.com/juletx/lm-evaluation-harness"},{"title":"national-center-for-ai-saudi-arabia/lm-evaluation-harness","url":"https://github.com/national-center-for-ai-saudi-arabia/lm-evaluation-harness"},{"title":"hilberthit/gpt-3","url":"https://github.com/hilberthit/gpt-3"},{"title":"longhao-chen/aicas2024","url":"https://github.com/longhao-chen/aicas2024"},{"title":"EightRice/atn_GPT-3","url":"https://github.com/EightRice/atn_GPT-3"},{"title":"Mind23-2/MindCode-138","url":"https://github.com/Mind23-2/MindCode-138"},{"title":"mbzuai-paris/lm-evaluation-harness-atlas-chat","url":"https://github.com/mbzuai-paris/lm-evaluation-harness-atlas-chat"},{"title":"Sypherd/lm-evaluation-harness","url":"https://github.com/Sypherd/lm-evaluation-harness"},{"title":"hojjat-mokhtarabadi/promptsource","url":"https://github.com/hojjat-mokhtarabadi/promptsource"},{"title":"zphang/lm_evaluation_harness","url":"https://github.com/zphang/lm_evaluation_harness"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/unsupervised-machine-translation-on-wmt2014-2","task":"Unsupervised Machine Translation","dataset_variant":"WMT2014 English-French","rows":7,"metrics":["BLEU"],"first_row_in_archive_order":{"model":"BERT-fused NMT","paper":"/paper/incorporating-bert-into-neural-machine-1","metrics":{"BLEU":"38.27"},"code_links":[{"title":"bert-nmt/bert-nmt","url":"https://github.com/bert-nmt/bert-nmt"},{"title":"vivekgohel56/Neural-machine-translation-english-to-polish","url":"https://github.com/vivekgohel56/Neural-machine-translation-english-to-polish"},{"title":"StuartCHAN/KARL","url":"https://github.com/StuartCHAN/KARL"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/machine-translation-on-wmt2014-french-english","task":"Machine Translation","dataset_variant":"WMT2014 French-English","rows":3,"metrics":["BLEU score"],"first_row_in_archive_order":{"model":"FLAN 137B (few-shot, k=9)","paper":"/paper/finetuned-language-models-are-zero-shot","metrics":{"BLEU score":"37.9"},"code_links":[{"title":"hiyouga/llama-efficient-tuning","url":"https://github.com/hiyouga/llama-efficient-tuning"},{"title":"bigcode-project/starcoder","url":"https://github.com/bigcode-project/starcoder"},{"title":"bigscience-workshop/promptsource","url":"https://github.com/bigscience-workshop/promptsource"},{"title":"google-research/flan","url":"https://github.com/google-research/flan"},{"title":"ukplab/arxiv2025-inherent-limits-plms","url":"https://github.com/ukplab/arxiv2025-inherent-limits-plms"},{"title":"openbiolink/promptsource","url":"https://github.com/openbiolink/promptsource"},{"title":"MS-P3/code6","url":"https://github.com/MS-P3/code6/tree/main/finetune"},{"title":"hojjat-mokhtarabadi/promptsource","url":"https://github.com/hojjat-mokhtarabadi/promptsource"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/machine-translation-on-wmt2014-english-czech","task":"Machine Translation","dataset_variant":"WMT2014 English-Czech","rows":2,"metrics":["BLEU score"],"first_row_in_archive_order":{"model":"Evolved Transformer Big","paper":"/paper/the-evolved-transformer","metrics":{"BLEU score":"28.2"},"code_links":[{"title":"tensorflow/tensor2tensor","url":"https://github.com/tensorflow/tensor2tensor"},{"title":"nazarov-yuriy/zh-ru-shared-task","url":"https://github.com/nazarov-yuriy/zh-ru-shared-task"},{"title":"moon23k/Transformer_Archs","url":"https://github.com/moon23k/Transformer_Archs"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/unsupervised-machine-translation-on-wmt2014","task":"Unsupervised Machine Translation","dataset_variant":"WMT2014 English-German","rows":2,"metrics":["BLEU"],"first_row_in_archive_order":{"model":"SMT + NMT (tuning and joint refinement)","paper":"/paper/an-effective-approach-to-unsupervised-machine","metrics":{"BLEU":"22.5"},"code_links":[{"title":"artetxem/monoses","url":"https://github.com/artetxem/monoses"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/unsupervised-machine-translation-on-wmt2014-3","task":"Unsupervised Machine Translation","dataset_variant":"WMT2014 German-English","rows":2,"metrics":["BLEU"],"first_row_in_archive_order":{"model":"SMT + NMT (tuning and joint refinement)","paper":"/paper/an-effective-approach-to-unsupervised-machine","metrics":{"BLEU":"27.0"},"code_links":[{"title":"artetxem/monoses","url":"https://github.com/artetxem/monoses"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/partialformer-modeling-part-instead-of-whole","title":"PartialFormer: Modeling Part Instead of Whole for Machine Translation","date":"2023-10-23","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/mega-moving-average-equipped-gated-attention","title":"Mega: Moving Average Equipped Gated Attention","date":"2022-09-21","rows_on_this_dataset":2,"code_links":7,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":12,"samples_unverified":1,"pointer_only_for_licence":8,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/bi-simcut-a-simple-strategy-for-boosting-1","title":"Bi-SimCut: A Simple Strategy for Boosting Neural Machine Translation","date":"2022-06-06","rows_on_this_dataset":4,"code_links":1,"syntology":null},{"paper":"/paper/bert-mbert-or-bibert-a-study-on","title":"BERT, mBERT, or BiBERT? A Study on Contextualized Embeddings for Neural Machine Translation","date":"2021-09-09","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/finetuned-language-models-are-zero-shot","title":"Finetuned Language Models Are Zero-Shot Learners","date":"2021-09-03","rows_on_this_dataset":4,"code_links":8,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/r-drop-regularized-dropout-for-neural","title":"R-Drop: Regularized Dropout for Neural Networks","date":"2021-06-28","rows_on_this_dataset":2,"code_links":8,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":2,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/resmlp-feedforward-networks-for-image","title":"ResMLP: Feedforward networks for image classification with data-efficient training","date":"2021-05-07","rows_on_this_dataset":4,"code_links":19,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":2,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/lessons-on-parameter-sharing-across-layers-in","title":"Lessons on Parameter Sharing across Layers in Transformers","date":"2021-04-13","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/rethinking-perturbations-in-encoder-decoders","title":"Rethinking Perturbations in Encoder-Decoders for Fast Training","date":"2021-04-05","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mask-attention-networks-rethinking-and","title":"Mask Attention Networks: Rethinking and Strengthen Transformer","date":"2021-03-25","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":0,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/finetuning-pretrained-transformers-into-rnns","title":"Finetuning Pretrained Transformers into RNNs","date":"2021-03-24","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":3,"samples_unverified":5,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/non-autoregressive-translation-by-learning","title":"Non-Autoregressive Translation by Learning Target Categorical Codes","date":"2021-03-21","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/random-feature-attention-1","title":"Random Feature Attention","date":"2021-03-03","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/omninet-omnidirectional-representations-from","title":"OmniNet: Omnidirectional Representations from Transformers","date":"2021-03-01","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/autodropout-learning-dropout-patterns-to","title":"AutoDropout: Learning Dropout Patterns to Regularize Deep Networks","date":"2021-01-05","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/subformer-a-parameter-reduced-transformer","title":"Subformer: A Parameter Reduced Transformer","date":"2021-01-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/incorporating-a-local-translation-mechanism","title":"Incorporating a Local Translation Mechanism into Non-autoregressive Translation","date":"2020-11-12","rows_on_this_dataset":4,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":3,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/pre-training-multilingual-neural-machine","title":"Pre-training Multilingual Neural Machine Translation by Leveraging Alignment Information","date":"2020-10-07","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/very-deep-transformers-for-neural-machine","title":"Very Deep Transformers for Neural Machine Translation","date":"2020-08-18","rows_on_this_dataset":3,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":6,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/glancing-transformer-for-non-autoregressive","title":"Glancing Transformer for Non-Autoregressive Neural Machine Translation","date":"2020-08-18","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/advaug-robust-adversarial-augmentation-for-1","title":"AdvAug: Robust Adversarial Augmentation for Neural Machine Translation","date":"2020-06-21","rows_on_this_dataset":3,"code_links":0,"syntology":null},{"paper":"/paper/multi-branch-attentive-transformer","title":"Multi-branch Attentive Transformer","date":"2020-06-18","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/language-models-are-few-shot-learners","title":"Language Models are Few-Shot Learners","date":"2020-05-28","rows_on_this_dataset":2,"code_links":67,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":65,"samples_ran":15,"samples_unverified":50,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/hat-hardware-aware-transformers-for-efficient","title":"HAT: Hardware-Aware Transformers for Efficient Natural Language Processing","date":"2020-05-28","rows_on_this_dataset":2,"code_links":4,"syntology":null},{"paper":"/paper/synthesizer-rethinking-self-attention-in","title":"Synthesizer: Rethinking Self-Attention in Transformer Models","date":"2020-05-02","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/lite-transformer-with-long-short-range","title":"Lite Transformer with Long-Short Range Attention","date":"2020-04-24","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/understanding-the-difficulty-of-training","title":"Understanding the Difficulty of Training Transformers","date":"2020-04-17","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":4,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/rethinking-batch-normalization-in","title":"PowerNorm: Rethinking Batch Normalization in Transformers","date":"2020-03-17","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/learning-to-encode-position-for-transformer","title":"Learning to Encode Position for Transformer with Continuous Dynamical Model","date":"2020-03-13","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":3,"samples_unverified":3,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/wide-minima-density-hypothesis-and-the","title":"Wide-minima Density Hypothesis and the Explore-Exploit Learning Rate Schedule","date":"2020-03-09","rows_on_this_dataset":1,"code_links":5,"syntology":null},{"paper":"/paper/incorporating-bert-into-neural-machine-1","title":"Incorporating BERT into Neural Machine Translation","date":"2020-02-17","rows_on_this_dataset":3,"code_links":3,"syntology":null},{"paper":"/paper/time-aware-large-kernel-convolutions","title":"Time-aware Large Kernel Convolutions","date":"2020-02-08","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":5,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/non-autoregressive-translation-with","title":"Non-autoregressive Translation with Disentangled Context Transformer","date":"2020-01-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/muse-parallel-multi-scale-attention-for","title":"MUSE: Parallel Multi-Scale Attention for Sequence to Sequence Learning","date":"2019-11-17","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/data-diversification-an-elegant-strategy-for","title":"Data Diversification: A Simple Strategy For Neural Machine Translation","date":"2019-11-05","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/exploring-the-limits-of-transfer-learning","title":"Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer","date":"2019-10-23","rows_on_this_dataset":2,"code_links":57,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":31,"samples_ran":2,"samples_unverified":29,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/flowseq-non-autoregressive-conditional","title":"FlowSeq: Non-Autoregressive Conditional Sequence Generation with Generative Flow","date":"2019-09-05","rows_on_this_dataset":10,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/adaptively-sparse-transformers","title":"Adaptively Sparse Transformers","date":"2019-08-30","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/depth-growing-for-neural-machine-translation","title":"Depth Growing for Neural Machine Translation","date":"2019-07-03","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/improving-neural-language-modeling-via","title":"Improving Neural Language Modeling via Adversarial Training","date":"2019-06-10","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/kermit-generative-insertion-based-modeling","title":"KERMIT: Generative Insertion-Based Modeling for Sequences","date":"2019-06-04","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/levenshtein-transformer","title":"Levenshtein Transformer","date":"2019-05-27","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/190506596","title":"Joint Source-Target Self Attention with Locality Constraints","date":"2019-05-16","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/deep-residual-output-layers-for-neural","title":"Deep Residual Output Layers for Neural Language Generation","date":"2019-05-14","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/synchronous-bidirectional-neural-machine","title":"Synchronous Bidirectional Neural Machine Translation","date":"2019-05-13","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/mass-masked-sequence-to-sequence-pre-training","title":"MASS: Masked Sequence to Sequence Pre-training for Language Generation","date":"2019-05-07","rows_on_this_dataset":2,"code_links":7,"syntology":null},{"paper":"/paper/an-effective-approach-to-unsupervised-machine","title":"An Effective Approach to Unsupervised Machine Translation","date":"2019-02-04","rows_on_this_dataset":4,"code_links":1,"syntology":null},{"paper":"/paper/the-evolved-transformer","title":"The Evolved Transformer","date":"2019-01-30","rows_on_this_dataset":6,"code_links":3,"syntology":null},{"paper":"/paper/memory-efficient-adaptive-optimization-for","title":"Memory-Efficient Adaptive Optimization","date":"2019-01-30","rows_on_this_dataset":1,"code_links":4,"syntology":null},{"paper":"/paper/pay-less-attention-with-lightweight-and","title":"Pay Less Attention with Lightweight and Dynamic Convolutions","date":"2019-01-29","rows_on_this_dataset":4,"code_links":3,"syntology":null},{"paper":"/paper/cross-lingual-language-model-pretraining","title":"Cross-lingual Language Model Pretraining","date":"2019-01-22","rows_on_this_dataset":2,"code_links":17,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":1,"samples_unverified":6,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/unsupervised-neural-machine-translation-with","title":"Unsupervised Neural Machine Translation with SMT as Posterior Regularization","date":"2019-01-14","rows_on_this_dataset":4,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":0,"samples_unverified":15,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/neural-machine-translation-with-adequacy","title":"Neural Machine Translation with Adequacy-Oriented Learning","date":"2018-11-21","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/modeling-localness-for-self-attention","title":"Modeling Localness for Self-Attention Networks","date":"2018-10-24","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/fast-and-simple-mixture-of-softmaxes-with-bpe","title":"Fast and Simple Mixture of Softmaxes with BPE and Hybrid-LightRNN for Language Generation","date":"2018-09-25","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/frage-frequency-agnostic-word-representation","title":"FRAGE: Frequency-Agnostic Word Representation","date":"2018-09-18","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/unsupervised-statistical-machine-translation","title":"Unsupervised Statistical Machine Translation","date":"2018-09-04","rows_on_this_dataset":5,"code_links":3,"syntology":null},{"paper":"/paper/understanding-back-translation-at-scale","title":"Understanding Back-Translation at Scale","date":"2018-08-28","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/universal-transformers","title":"Universal Transformers","date":"2018-07-10","rows_on_this_dataset":1,"code_links":8,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":25,"samples_ran":14,"samples_unverified":11,"pointer_only_for_licence":24,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/dense-information-flow-for-neural-machine","title":"Dense Information Flow for Neural Machine Translation","date":"2018-06-03","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/scaling-neural-machine-translation","title":"Scaling Neural Machine Translation","date":"2018-06-01","rows_on_this_dataset":2,"code_links":5,"syntology":null},{"paper":"/paper/accelerating-neural-transformer-via-an","title":"Accelerating Neural Transformer via an Average Attention Network","date":"2018-05-02","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":19,"samples_ran":0,"samples_unverified":19,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/the-best-of-both-worlds-combining-recent","title":"The Best of Both Worlds: Combining Recent Advances in Neural Machine Translation","date":"2018-04-26","rows_on_this_dataset":2,"code_links":3,"syntology":null},{"paper":"/paper/phrase-based-neural-unsupervised-machine","title":"Phrase-Based & Neural Unsupervised Machine Translation","date":"2018-04-20","rows_on_this_dataset":8,"code_links":14,"syntology":null},{"paper":"/paper/self-attention-with-relative-position","title":"Self-Attention with Relative Position Representations","date":"2018-03-06","rows_on_this_dataset":2,"code_links":13,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":23,"samples_ran":13,"samples_unverified":10,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/deterministic-non-autoregressive-neural","title":"Deterministic Non-Autoregressive Neural Sequence Modeling by Iterative Refinement","date":"2018-02-19","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/deliberation-networks-sequence-generation","title":"Deliberation Networks: Sequence Generation Beyond One-Pass Decoding","date":"2017-12-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/non-autoregressive-neural-machine-translation-1","title":"Non-Autoregressive Neural Machine Translation","date":"2017-11-07","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/weighted-transformer-network-for-machine","title":"Weighted Transformer Network for Machine Translation","date":"2017-11-06","rows_on_this_dataset":2,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/unsupervised-neural-machine-translation","title":"Unsupervised Neural Machine Translation","date":"2017-10-30","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":6,"samples_unverified":0,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/simple-recurrent-units-for-highly","title":"Simple Recurrent Units for Highly Parallelizable Recurrence","date":"2017-09-08","rows_on_this_dataset":1,"code_links":11,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/attention-is-all-you-need","title":"Attention Is All You Need","date":"2017-06-12","rows_on_this_dataset":4,"code_links":595,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":946,"samples_ran":600,"samples_unverified":346,"pointer_only_for_licence":451,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/depthwise-separable-convolutions-for-neural","title":"Depthwise Separable Convolutions for Neural Machine Translation","date":"2017-06-09","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/convolutional-sequence-to-sequence-learning","title":"Convolutional Sequence to Sequence Learning","date":"2017-05-08","rows_on_this_dataset":4,"code_links":37,"syntology":null},{"paper":"/paper/outrageously-large-neural-networks-the","title":"Outrageously Large Neural Networks: The Sparsely-Gated Mixture-of-Experts Layer","date":"2017-01-23","rows_on_this_dataset":2,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":4,"samples_unverified":2,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/a-convolutional-encoder-model-for-neural","title":"A Convolutional Encoder Model for Neural Machine Translation","date":"2016-11-07","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/neural-machine-translation-in-linear-time","title":"Neural Machine Translation in Linear Time","date":"2016-10-31","rows_on_this_dataset":1,"code_links":11,"syntology":null},{"paper":"/paper/can-active-memory-replace-attention","title":"Can Active Memory Replace Attention?","date":"2016-10-27","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/googles-neural-machine-translation-system","title":"Google's Neural Machine Translation System: Bridging the Gap between Human and Machine Translation","date":"2016-09-26","rows_on_this_dataset":2,"code_links":28,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":46,"samples_ran":23,"samples_unverified":23,"pointer_only_for_licence":12,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/neural-semantic-encoders","title":"Neural Semantic Encoders","date":"2016-07-14","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/sequence-level-knowledge-distillation","title":"Sequence-Level Knowledge Distillation","date":"2016-06-25","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/deep-recurrent-models-with-fast-forward","title":"Deep Recurrent Models with Fast-Forward Connections for Neural Machine Translation","date":"2016-06-14","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/edinburghs-syntax-based-systems-at-wmt-2015","title":"Edinburgh's Syntax-Based Systems at WMT 2015","date":"2015-09-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/effective-approaches-to-attention-based","title":"Effective Approaches to Attention-based Neural Machine Translation","date":"2015-08-17","rows_on_this_dataset":3,"code_links":44,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":2,"samples_unverified":6,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/addressing-the-rare-word-problem-in-neural","title":"Addressing the Rare Word Problem in Neural Machine Translation","date":"2014-10-30","rows_on_this_dataset":1,"code_links":5,"syntology":null},{"paper":"/paper/sequence-to-sequence-learning-with-neural","title":"Sequence to Sequence Learning with Neural Networks","date":"2014-09-10","rows_on_this_dataset":2,"code_links":74,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":25,"samples_ran":11,"samples_unverified":14,"pointer_only_for_licence":9,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/recurrent-neural-network-regularization","title":"Recurrent Neural Network Regularization","date":"2014-09-08","rows_on_this_dataset":1,"code_links":21,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":2,"samples_unverified":4,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/neural-machine-translation-by-jointly","title":"Neural Machine Translation by Jointly Learning to Align and Translate","date":"2014-09-01","rows_on_this_dataset":1,"code_links":124,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":44,"samples_ran":21,"samples_unverified":23,"pointer_only_for_licence":16,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/learning-phrase-representations-using-rnn","title":"Learning Phrase Representations using RNN Encoder-Decoder for Statistical Machine Translation","date":"2014-06-03","rows_on_this_dataset":1,"code_links":42,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":22,"samples_ran":10,"samples_unverified":12,"pointer_only_for_licence":14,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":42,"samples_harvested":1394,"samples_ran":788,"samples_unverified":606,"pointer_only_for_licence":587,"papers_with_no_sample_that_ran":6,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}