{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/stacked-attention-networks-for-image-question","title":"Stacked Attention Networks for Image Question Answering","arxiv_id":"1511.02274","date":"2015-11-07","proceeding":"CVPR 2016 6","authors":["Zichao Yang","Xiaodong He","Jianfeng Gao","Li Deng","Alex Smola"],"abstract":"This paper presents stacked attention networks (SANs) that learn to answer\nnatural language questions from images. SANs use semantic representation of a\nquestion as query to search for the regions in an image that are related to the\nanswer. We argue that image question answering (QA) often requires multiple\nsteps of reasoning. Thus, we develop a multiple-layer SAN in which we query an\nimage multiple times to infer the answer progressively. Experiments conducted\non four image QA data sets demonstrate that the proposed SANs significantly\noutperform previous state-of-the-art approaches. The visualization of the\nattention layers illustrates the progress that the SAN locates the relevant\nvisual clues that lead to the answer of the question layer-by-layer.","url_abs":"http://arxiv.org/abs/1511.02274v2","url_pdf":"http://arxiv.org/pdf/1511.02274v2.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"stacked-attention-networks-for-image-question","repo_url":"https://github.com/Cold-Winter/vqs","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"caffe2","reach":{"status":"unanswered"}},{"paper_slug":"stacked-attention-networks-for-image-question","repo_url":"https://github.com/SatyamGaba/visual_question_answering","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"unanswered"}},{"paper_slug":"stacked-attention-networks-for-image-question","repo_url":"https://github.com/SatyamGaba/vqa","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"unanswered"}},{"paper_slug":"stacked-attention-networks-for-image-question","repo_url":"https://github.com/Shivanshu-Gupta/Visual-Question-Answering","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"unanswered"}},{"paper_slug":"stacked-attention-networks-for-image-question","repo_url":"https://github.com/TingAnChien/san-vqa-tensorflow","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"ok"}},{"paper_slug":"stacked-attention-networks-for-image-question","repo_url":"https://github.com/abhi-iyer/visual-question-answering","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok"}},{"paper_slug":"stacked-attention-networks-for-image-question","repo_url":"https://github.com/abhijit-buet/VizWiz-Visua-Question-Answering-2021","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"unanswered"}},{"paper_slug":"stacked-attention-networks-for-image-question","repo_url":"https://github.com/abhijit-buet/VizWiz-Visual-Question-Answering-2021","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"stacked-attention-networks-for-image-question","repo_url":"https://github.com/abhshkdz/neural-vqa-attention","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"torch","reach":{"status":"ok"}},{"paper_slug":"stacked-attention-networks-for-image-question","repo_url":"https://github.com/chirag26495/DAN_VQA","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"stacked-attention-networks-for-image-question","repo_url":"https://github.com/jiayi-wei/vqa-tf2","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"ok"}},{"paper_slug":"stacked-attention-networks-for-image-question","repo_url":"https://github.com/mokhalid-dev/Attention-based-VQA-model","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"unanswered"}},{"paper_slug":"stacked-attention-networks-for-image-question","repo_url":"https://github.com/rs9000/VisualReasoning_MMnet","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"unanswered"}},{"paper_slug":"stacked-attention-networks-for-image-question","repo_url":"https://github.com/snagiri/ECE285_Jarvis_ProjectA","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"unanswered"}},{"paper_slug":"stacked-attention-networks-for-image-question","repo_url":"https://github.com/yanxinyan1/yxy","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"unanswered"}},{"paper_slug":"stacked-attention-networks-for-image-question","repo_url":"https://github.com/zcyang/imageqa-san","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":{"status":"ok"}}],"tasks":[{"task_slug":"visual-question-answering","task_name":"Visual Question Answering (VQA)"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[{"leaderboard":"/sota/visual-question-answering-on-coco-visual-4","task":"Visual Question Answering (VQA)","dataset":"COCO Visual Question Answering (VQA) real images 1.0 open ended","model":"SAN","rank_in_archive_order":11,"of":14,"metrics":{"Percentage correct":"58.9"},"uses_additional_data":false},{"leaderboard":"/sota/visual-question-answering-on-vqa-v1-test-std","task":"Visual Question Answering (VQA)","dataset":"VQA v1 test-std","model":"SAN (VGG)","rank_in_archive_order":5,"of":6,"metrics":{"Accuracy":"58.9"},"uses_additional_data":false}],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=1511.02274","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1511.02274"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/zcyang/imageqa-san","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/abhijit-buet/VizWiz-Visual-Question-Answering-2021","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/chirag26495/DAN_VQA","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/mokhalid-dev/Attention-based-VQA-model","reach":{"status":"unanswered"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/Cold-Winter/vqs","reach":{"status":"unanswered"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/yanxinyan1/yxy","reach":{"status":"unanswered"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/abhi-iyer/visual-question-answering","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/SatyamGaba/vqa","reach":{"status":"unanswered"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/SatyamGaba/visual_question_answering","reach":{"status":"unanswered"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/snagiri/ECE285_Jarvis_ProjectA","reach":{"status":"unanswered"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/Shivanshu-Gupta/Visual-Question-Answering","reach":{"status":"unanswered"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/TingAnChien/san-vqa-tensorflow","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/abhijit-buet/VizWiz-Visua-Question-Answering-2021","reach":{"status":"unanswered"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/rs9000/VisualReasoning_MMnet","reach":{"status":"unanswered"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/jiayi-wei/vqa-tf2","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/abhshkdz/neural-vqa-attention","reach":{"status":"ok"}}],"summary":{"ran_draft_wrong":2,"ran_fixture":2},"by_repo_kind":{"listed":{"samples":4,"ran":4,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":4,"samples":[{"code_sha256_prefix":"388d2a6ba37219cd","entry":"apply_attention","repo":"abhijit-buet/VizWiz-Visual-Question-Answering-2021","repo_kind":"listed","path":"models.py","file_url":"https://github.com/abhijit-buet/VizWiz-Visual-Question-Answering-2021/blob/HEAD/models.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"388d2a6ba37219cd"}},{"code_sha256_prefix":"d12f1716d21540a8","entry":"create_submission","repo":"abhijit-buet/VizWiz-Visual-Question-Answering-2021","repo_kind":"listed","path":"predict.py","file_url":"https://github.com/abhijit-buet/VizWiz-Visual-Question-Answering-2021/blob/HEAD/predict.py","link_basis":"first_harvest_node","language":"python","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"d12f1716d21540a8"}},{"code_sha256_prefix":"6d3a7a351f38eccd","entry":"predict_answers","repo":"abhijit-buet/VizWiz-Visual-Question-Answering-2021","repo_kind":"listed","path":"predict.py","file_url":"https://github.com/abhijit-buet/VizWiz-Visual-Question-Answering-2021/blob/HEAD/predict.py","link_basis":"first_harvest_node","language":"python","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"6d3a7a351f38eccd"}},{"code_sha256_prefix":"fe3010862617673c","entry":"repeat_encoded_question","repo":"abhijit-buet/VizWiz-Visual-Question-Answering-2021","repo_kind":"listed","path":"models.py","file_url":"https://github.com/abhijit-buet/VizWiz-Visual-Question-Answering-2021/blob/HEAD/models.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"fe3010862617673c"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}