{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/doubly-attentive-decoder-for-multi-modal","title":"Doubly-Attentive Decoder for Multi-modal Neural Machine Translation","arxiv_id":"1702.01287","date":"2017-02-04","proceeding":"ACL 2017 7","authors":["Iacer Calixto","Qun Liu","Nick Campbell"],"abstract":"We introduce a Multi-modal Neural Machine Translation model in which a\ndoubly-attentive decoder naturally incorporates spatial visual features\nobtained using pre-trained convolutional neural networks, bridging the gap\nbetween image description and translation. Our decoder learns to attend to\nsource-language words and parts of an image independently by means of two\nseparate attention mechanisms as it generates words in the target language. We\nfind that our model can efficiently exploit not just back-translated in-domain\nmulti-modal data but also large general-domain text-only MT corpora. We also\nreport state-of-the-art results on the Multi30k data set.","url_abs":"http://arxiv.org/abs/1702.01287v1","url_pdf":"http://arxiv.org/pdf/1702.01287v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[],"tasks":[{"task_slug":"decoder","task_name":"Decoder"},{"task_slug":null,"task_name":"Image Description"},{"task_slug":"machine-translation","task_name":"Machine Translation"},{"task_slug":"multimodal-machine-translation","task_name":"Multimodal Machine Translation"},{"task_slug":"translation","task_name":"Translation"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[{"leaderboard":"/sota/multimodal-machine-translation-on-multi30k","task":"Multimodal Machine Translation","dataset":"Multi30K","model":"NMTSRC+IMG","rank_in_archive_order":11,"of":15,"metrics":{"BLEU (EN-DE)":"37.1","Meteor (EN-DE)":"54.5"},"uses_additional_data":false}],"syntology":{"syntology_url":"https://syntology.ai/paper/1702.01287","atlas_url":"https://app.syntology.ai/?focus=1702.01287","mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}