{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/learning-deep-representations-of-fine-grained","title":"Learning Deep Representations of Fine-grained Visual Descriptions","arxiv_id":"1605.05395","date":"2016-05-17","proceeding":"CVPR 2016 6","authors":["Scott Reed","Zeynep Akata","Bernt Schiele","Honglak Lee"],"abstract":"State-of-the-art methods for zero-shot visual recognition formulate learning\nas a joint embedding problem of images and side information. In these\nformulations the current best complement to visual features are attributes:\nmanually encoded vectors describing shared characteristics among categories.\nDespite good performance, attributes have limitations: (1) finer-grained\nrecognition requires commensurately more attributes, and (2) attributes do not\nprovide a natural language interface. We propose to overcome these limitations\nby training neural language models from scratch; i.e. without pre-training and\nonly consuming words and characters. Our proposed models train end-to-end to\nalign with the fine-grained and category-specific content of images. Natural\nlanguage provides a flexible and compact way of encoding only the salient\nvisual aspects for distinguishing categories. By training on raw text, our\nmodel can do inference on raw text as well, providing humans a familiar mode\nboth for annotation and retrieval. Our model achieves strong performance on\nzero-shot text-based image retrieval and significantly outperforms the\nattribute-based state-of-the-art for zero-shot classification on the Caltech\nUCSD Birds 200-2011 dataset.","url_abs":"http://arxiv.org/abs/1605.05395v1","url_pdf":"http://arxiv.org/pdf/1605.05395v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"learning-deep-representations-of-fine-grained","repo_url":"https://github.com/Maymaher/StackGANv2","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"unanswered"}},{"paper_slug":"learning-deep-representations-of-fine-grained","repo_url":"https://github.com/Vigneshthanga/stackGAN-v2","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"unanswered"}},{"paper_slug":"learning-deep-representations-of-fine-grained","repo_url":"https://github.com/Vishal-V/StackGAN","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"learning-deep-representations-of-fine-grained","repo_url":"https://github.com/hanzhanggit/StackGAN-inception-model","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":null},{"paper_slug":"learning-deep-representations-of-fine-grained","repo_url":"https://github.com/hanzhanggit/StackGAN-v2","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"learning-deep-representations-of-fine-grained","repo_url":"https://github.com/priscillalui/StackGAN-Stories","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"learning-deep-representations-of-fine-grained","repo_url":"https://github.com/rafiahmed40/stack-adverserial-network","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"learning-deep-representations-of-fine-grained","repo_url":"https://github.com/reedscot/cvpr2016","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"learning-deep-representations-of-fine-grained","repo_url":"https://github.com/rightlit/StackGAN-v2-rev","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}}],"tasks":[{"task_slug":"attribute","task_name":"Attribute"},{"task_slug":"image-retrieval","task_name":"Image Retrieval"},{"task_slug":"retrieval","task_name":"Retrieval"},{"task_slug":"zero-shot-learning","task_name":"Zero-Shot Learning"},{"task_slug":null,"task_name":"zero-shot-classification"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[{"leaderboard":"/sota/few-shot-image-classification-on-cub-200-50","task":"Few-Shot Image Classification","dataset":"CUB 200 50-way (0-shot)","model":"DA-SJE Reed et al. (2016)","rank_in_archive_order":2,"of":4,"metrics":{"Accuracy":"50.9"},"uses_additional_data":false},{"leaderboard":"/sota/few-shot-image-classification-on-cub-200-50","task":"Few-Shot Image Classification","dataset":"CUB 200 50-way (0-shot)","model":"DS-SJE Reed et al. (2016)","rank_in_archive_order":3,"of":4,"metrics":{"Accuracy":"50.4"},"uses_additional_data":false},{"leaderboard":"/sota/few-shot-image-classification-on-cub-200-2011-1","task":"Few-Shot Image Classification","dataset":"CUB-200-2011 - 0-Shot","model":"Word CNN-RNN (DS-SJE Embedding)","rank_in_archive_order":1,"of":5,"metrics":{"AP50":"48.7","Top-1 Accuracy":"56.8%"},"uses_additional_data":false},{"leaderboard":"/sota/few-shot-image-classification-on-flowers-102-1","task":"Few-Shot Image Classification","dataset":"Flowers-102 - 0-Shot","model":"Word CNN-RNN (DS-SJE Embedding)","rank_in_archive_order":1,"of":1,"metrics":{"AP50":"59.6","Accuracy":"65.6%"},"uses_additional_data":false}],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=1605.05395","mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}