{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/language-model-prior-for-low-resource-neural","title":"Language Model Prior for Low-Resource Neural Machine Translation","arxiv_id":"2004.14928","date":"2020-04-30","proceeding":"EMNLP 2020 11","authors":["Christos Baziotis","Barry Haddow","Alexandra Birch"],"abstract":"The scarcity of large parallel corpora is an important obstacle for neural machine translation. A common solution is to exploit the knowledge of language models (LM) trained on abundant monolingual data. In this work, we propose a novel approach to incorporate a LM as prior in a neural translation model (TM). Specifically, we add a regularization term, which pushes the output distributions of the TM to be probable under the LM prior, while avoiding wrong predictions when the TM \"disagrees\" with the LM. This objective relates to knowledge distillation, where the LM can be viewed as teaching the TM about the target language. The proposed approach does not compromise decoding speed, because the LM is used only at training time, unlike previous work that requires it during inference. We present an analysis of the effects that different methods have on the distributions of the TM. Results on two low-resource machine translation datasets show clear improvements even with limited monolingual data.","url_abs":"https://arxiv.org/abs/2004.14928v3","url_pdf":"https://arxiv.org/pdf/2004.14928v3.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"language-model-prior-for-low-resource-neural","repo_url":"https://github.com/cbaziotis/lm-prior-for-nmt","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}}],"tasks":[{"task_slug":"knowledge-distillation","task_name":"Knowledge Distillation"},{"task_slug":"language-modeling","task_name":"Language Modeling"},{"task_slug":"language-modelling","task_name":"Language Modelling"},{"task_slug":"low-resource-neural-machine-translation-2","task_name":"Low Resource Neural Machine Translation"},{"task_slug":"low-resource-neural-machine-translation","task_name":"Low-Resource Neural Machine Translation"},{"task_slug":"machine-translation","task_name":"Machine Translation"},{"task_slug":"translation","task_name":"Translation"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"syntology_url":"https://syntology.ai/paper/2004.14928","atlas_url":"https://app.syntology.ai/?focus=2004.14928","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.14928"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-25T09:33:49+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/cbaziotis/lm-prior-for-nmt","reach":{"status":"ok","spdx":"MIT"}}],"summary":{"ran":3,"unverified":3},"by_repo_kind":{"official":{"samples":6,"ran":3,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":0,"samples":[{"code_sha256_prefix":"de1492b5da3b7bb5","entry":"baseline","repo":"cbaziotis/lm-prior-for-nmt","repo_kind":"official","path":"configs/transformer/generate_nmt_experiments.py","file_url":"https://github.com/cbaziotis/lm-prior-for-nmt/blob/HEAD/configs/transformer/generate_nmt_experiments.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"de1492b5da3b7bb5"}},{"code_sha256_prefix":"0471dc0474e44f8d","entry":"fusion","repo":"cbaziotis/lm-prior-for-nmt","repo_kind":"official","path":"configs/transformer/generate_nmt_experiments.py","file_url":"https://github.com/cbaziotis/lm-prior-for-nmt/blob/HEAD/configs/transformer/generate_nmt_experiments.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"0471dc0474e44f8d"}},{"code_sha256_prefix":"cb3d7ac764f53331","entry":"get_name","repo":"cbaziotis/lm-prior-for-nmt","repo_kind":"official","path":"configs/rnn/generate_nmt_experiments.py","file_url":"https://github.com/cbaziotis/lm-prior-for-nmt/blob/HEAD/configs/rnn/generate_nmt_experiments.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"cb3d7ac764f53331"}},{"code_sha256_prefix":"70ea29561d623be2","entry":"baseline","repo":"cbaziotis/lm-prior-for-nmt","repo_kind":"official","path":"configs/rnn/generate_nmt_experiments.py","file_url":"https://github.com/cbaziotis/lm-prior-for-nmt/blob/HEAD/configs/rnn/generate_nmt_experiments.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"70ea29561d623be2"}},{"code_sha256_prefix":"baf87cd1b3bd046c","entry":"fusion","repo":"cbaziotis/lm-prior-for-nmt","repo_kind":"official","path":"configs/rnn/generate_nmt_experiments.py","file_url":"https://github.com/cbaziotis/lm-prior-for-nmt/blob/HEAD/configs/rnn/generate_nmt_experiments.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"baf87cd1b3bd046c"}},{"code_sha256_prefix":"0f1a2cddb04257e0","entry":"get_name","repo":"cbaziotis/lm-prior-for-nmt","repo_kind":"official","path":"configs/transformer/generate_nmt_experiments.py","file_url":"https://github.com/cbaziotis/lm-prior-for-nmt/blob/HEAD/configs/transformer/generate_nmt_experiments.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"0f1a2cddb04257e0"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}