{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/from-captions-to-visual-concepts-and-back","title":"From Captions to Visual Concepts and Back","arxiv_id":"1411.4952","date":"2014-11-18","proceeding":"CVPR 2015 6","authors":["Hao Fang","Saurabh Gupta","Forrest Iandola","Rupesh Srivastava","Li Deng","Piotr Dollár","Jianfeng Gao","Xiaodong He","Margaret Mitchell","John C. Platt","C. Lawrence Zitnick","Geoffrey Zweig"],"abstract":"This paper presents a novel approach for automatically generating image\ndescriptions: visual detectors, language models, and multimodal similarity\nmodels learnt directly from a dataset of image captions. We use multiple\ninstance learning to train visual detectors for words that commonly occur in\ncaptions, including many different parts of speech such as nouns, verbs, and\nadjectives. The word detector outputs serve as conditional inputs to a\nmaximum-entropy language model. The language model learns from a set of over\n400,000 image descriptions to capture the statistics of word usage. We capture\nglobal semantics by re-ranking caption candidates using sentence-level features\nand a deep multimodal similarity model. Our system is state-of-the-art on the\nofficial Microsoft COCO benchmark, producing a BLEU-4 score of 29.1%. When\nhuman judges compare the system captions to ones written by other people on our\nheld-out test set, the system captions have equal or better quality 34% of the\ntime.","url_abs":"http://arxiv.org/abs/1411.4952v3","url_pdf":"http://arxiv.org/pdf/1411.4952v3.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"from-captions-to-visual-concepts-and-back","repo_url":"https://github.com/s-gupta/visual-concepts","is_official":1,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"none","reach":{"status":"ok","spdx":"BSD-2-Clause"}}],"tasks":[{"task_slug":"image-captioning","task_name":"Image Captioning"},{"task_slug":"language-modeling","task_name":"Language Modeling"},{"task_slug":"language-modelling","task_name":"Language Modelling"},{"task_slug":"multiple-instance-learning","task_name":"Multiple Instance Learning"},{"task_slug":"re-ranking","task_name":"Re-Ranking"},{"task_slug":"sentence","task_name":"Sentence"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[{"leaderboard":"/sota/image-captioning-on-coco-captions","task":"Image Captioning","dataset":"COCO Captions","model":"From Captions to Visual Concepts and Back","rank_in_archive_order":33,"of":41,"metrics":{"BLEU-4":"25.7","METEOR":"23.6"},"uses_additional_data":false},{"leaderboard":"/sota/image-captioning-on-coco-captions-test","task":"Image Captioning","dataset":"COCO Captions test","model":"From Captions to Visual Concepts and Back","rank_in_archive_order":1,"of":2,"metrics":{"BLEU-4":"56.7","CIDEr":"92.5","METEOR":"33.1"},"uses_additional_data":false}],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=1411.4952","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1411.4952"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/s-gupta/visual-concepts","reach":{"status":"ok","spdx":"BSD-2-Clause"}}],"summary":{"unverified":6},"by_repo_kind":{"official":{"samples":6,"ran":0,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":0,"samples":[{"code_sha256_prefix":"d8a9e2b78edf5dea","entry":"calc_pr_ovr","repo":"s-gupta/visual-concepts","repo_kind":"official","path":"cap_eval_utils.py","file_url":"https://github.com/s-gupta/visual-concepts/blob/HEAD/cap_eval_utils.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-2-Clause","inline_ok":true,"mcp_get_code":{"code_sha256":"d8a9e2b78edf5dea"}},{"code_sha256_prefix":"b7ad54c47f3903eb","entry":"compute_precision_score_mapping","repo":"s-gupta/visual-concepts","repo_kind":"official","path":"cap_eval_utils.py","file_url":"https://github.com/s-gupta/visual-concepts/blob/HEAD/cap_eval_utils.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-2-Clause","inline_ok":true,"mcp_get_code":{"code_sha256":"b7ad54c47f3903eb"}},{"code_sha256_prefix":"c63e11043a0ae6b7","entry":"get_vocab","repo":"s-gupta/visual-concepts","repo_kind":"official","path":"preprocess.py","file_url":"https://github.com/s-gupta/visual-concepts/blob/HEAD/preprocess.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-2-Clause","inline_ok":true,"mcp_get_code":{"code_sha256":"c63e11043a0ae6b7"}},{"code_sha256_prefix":"b4592d0a1e0ef721","entry":"get_vocab_counts","repo":"s-gupta/visual-concepts","repo_kind":"official","path":"preprocess.py","file_url":"https://github.com/s-gupta/visual-concepts/blob/HEAD/preprocess.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-2-Clause","inline_ok":true,"mcp_get_code":{"code_sha256":"b4592d0a1e0ef721"}},{"code_sha256_prefix":"7f491790f87d4ca2","entry":"get_vocab_top_k","repo":"s-gupta/visual-concepts","repo_kind":"official","path":"preprocess.py","file_url":"https://github.com/s-gupta/visual-concepts/blob/HEAD/preprocess.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-2-Clause","inline_ok":true,"mcp_get_code":{"code_sha256":"7f491790f87d4ca2"}},{"code_sha256_prefix":"18581418250668ab","entry":"voc_ap","repo":"s-gupta/visual-concepts","repo_kind":"official","path":"cap_eval_utils.py","file_url":"https://github.com/s-gupta/visual-concepts/blob/HEAD/cap_eval_utils.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-2-Clause","inline_ok":true,"mcp_get_code":{"code_sha256":"18581418250668ab"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}