{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/mutan-multimodal-tucker-fusion-for-visual","title":"MUTAN: Multimodal Tucker Fusion for Visual Question Answering","arxiv_id":"1705.06676","date":"2017-05-18","proceeding":"ICCV 2017 10","authors":["Hedi Ben-Younes","Rémi Cadene","Matthieu Cord","Nicolas Thome"],"abstract":"Bilinear models provide an appealing framework for mixing and merging\ninformation in Visual Question Answering (VQA) tasks. They help to learn high\nlevel associations between question meaning and visual concepts in the image,\nbut they suffer from huge dimensionality issues. We introduce MUTAN, a\nmultimodal tensor-based Tucker decomposition to efficiently parametrize\nbilinear interactions between visual and textual representations. Additionally\nto the Tucker framework, we design a low-rank matrix-based decomposition to\nexplicitly constrain the interaction rank. With MUTAN, we control the\ncomplexity of the merging scheme while keeping nice interpretable fusion\nrelations. We show how our MUTAN model generalizes some of the latest VQA\narchitectures, providing state-of-the-art results.","url_abs":"http://arxiv.org/abs/1705.06676v1","url_pdf":"http://arxiv.org/pdf/1705.06676v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"mutan-multimodal-tucker-fusion-for-visual","repo_url":"https://github.com/Cadene/vqa.pytorch","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"unanswered"}},{"paper_slug":"mutan-multimodal-tucker-fusion-for-visual","repo_url":"https://github.com/Adam1679/mutan-article-net","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"unanswered"}},{"paper_slug":"mutan-multimodal-tucker-fusion-for-visual","repo_url":"https://github.com/Shivanshu-Gupta/Visual-Question-Answering","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"unanswered"}},{"paper_slug":"mutan-multimodal-tucker-fusion-for-visual","repo_url":"https://github.com/gabegrand/adversarial-vqa","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"mutan-multimodal-tucker-fusion-for-visual","repo_url":"https://github.com/vuhoangminh/vqa_medical","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"unanswered"}},{"paper_slug":"mutan-multimodal-tucker-fusion-for-visual","repo_url":"https://github.com/yikang-li/iqan","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"unanswered"}}],"tasks":[{"task_slug":"visual-question-answering-1","task_name":"Visual Question Answering"},{"task_slug":"visual-question-answering","task_name":"Visual Question Answering (VQA)"}],"methods":[{"method_slug":"affine-coupling","method_name":"Affine Coupling"},{"method_slug":"normalizing-flows","method_name":"Normalizing Flows"}],"datasets_introduced":[],"methods_introduced":[],"results":[{"leaderboard":"/sota/visual-question-answering-on-vqa-v2-test-dev","task":"Visual Question Answering (VQA)","dataset":"VQA v2 test-dev","model":"MUTAN","rank_in_archive_order":39,"of":56,"metrics":{"Accuracy":"67.42"},"uses_additional_data":false},{"leaderboard":"/sota/visual-question-answering-on-vqa-v2-test-std","task":"Visual Question Answering (VQA)","dataset":"VQA v2 test-std","model":"MUTAN","rank_in_archive_order":34,"of":38,"metrics":{"overall":"67.4"},"uses_additional_data":false}],"syntology":{"syntology_url":"https://syntology.ai/paper/1705.06676","atlas_url":"https://app.syntology.ai/?focus=1705.06676","mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}