{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/generation-and-comprehension-of-unambiguous","title":"Generation and Comprehension of Unambiguous Object Descriptions","arxiv_id":"1511.02283","date":"2015-11-07","proceeding":"CVPR 2016 6","authors":["Junhua Mao","Jonathan Huang","Alexander Toshev","Oana Camburu","Alan Yuille","Kevin Murphy"],"abstract":"We propose a method that can generate an unambiguous description (known as a\nreferring expression) of a specific object or region in an image, and which can\nalso comprehend or interpret such an expression to infer which object is being\ndescribed. We show that our method outperforms previous methods that generate\ndescriptions of objects without taking into account other potentially ambiguous\nobjects in the scene. Our model is inspired by recent successes of deep\nlearning methods for image captioning, but while image captioning is difficult\nto evaluate, our task allows for easy objective evaluation. We also present a\nnew large-scale dataset for referring expressions, based on MS-COCO. We have\nreleased the dataset and a toolbox for visualization and evaluation, see\nhttps://github.com/mjhucla/Google_Refexp_toolbox","url_abs":"http://arxiv.org/abs/1511.02283v3","url_pdf":"http://arxiv.org/pdf/1511.02283v3.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"generation-and-comprehension-of-unambiguous","repo_url":"https://github.com/mjhucla/Google_Refexp_toolbox","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"none","reach":{"status":"ok"}}],"tasks":[{"task_slug":"image-captioning","task_name":"Image Captioning"},{"task_slug":"object","task_name":"Object"},{"task_slug":"referring-expression","task_name":"Referring Expression"}],"methods":[],"datasets_introduced":[{"slug":"google-refexp","name":"Google Refexp","full_name":""}],"methods_introduced":[],"results":[],"syntology":{"syntology_url":"https://syntology.ai/paper/1511.02283","atlas_url":"https://app.syntology.ai/?focus=1511.02283","mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}