{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/vinvl-making-visual-representations-matter-in","title":"VinVL: Revisiting Visual Representations in Vision-Language Models","arxiv_id":"2101.00529","date":"2021-01-02","proceeding":"CVPR 2021 1","authors":["Pengchuan Zhang","Xiujun Li","Xiaowei Hu","Jianwei Yang","Lei Zhang","Lijuan Wang","Yejin Choi","Jianfeng Gao"],"abstract":"This paper presents a detailed study of improving visual representations for vision language (VL) tasks and develops an improved object detection model to provide object-centric representations of images. Compared to the most widely used \\emph{bottom-up and top-down} model \\cite{anderson2018bottom}, the new model is bigger, better-designed for VL tasks, and pre-trained on much larger training corpora that combine multiple public annotated object detection datasets. Therefore, it can generate representations of a richer collection of visual objects and concepts. While previous VL research focuses mainly on improving the vision-language fusion model and leaves the object detection model improvement untouched, we show that visual features matter significantly in VL models. In our experiments we feed the visual features generated by the new object detection model into a Transformer-based VL fusion model \\oscar \\cite{li2020oscar}, and utilize an improved approach \\short\\ to pre-train the VL model and fine-tune it on a wide range of downstream VL tasks. Our results show that the new visual features significantly improve the performance across all VL tasks, creating new state-of-the-art results on seven public benchmarks. We will release the new object detection model to public.","url_abs":"https://arxiv.org/abs/2101.00529v2","url_pdf":"https://arxiv.org/pdf/2101.00529v2.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"vinvl-making-visual-representations-matter-in","repo_url":"https://github.com/pzzhang/VinVL","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"none","reach":{"status":"unanswered"}},{"paper_slug":"vinvl-making-visual-representations-matter-in","repo_url":"https://github.com/JoshuaPlacidi/MS-COCO-Object-Tags","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":{"status":"unanswered"}},{"paper_slug":"vinvl-making-visual-representations-matter-in","repo_url":"https://github.com/JoshuaPlacidi/MS-COCO-Tags","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":{"status":"unanswered"}},{"paper_slug":"vinvl-making-visual-representations-matter-in","repo_url":"https://github.com/cattidea/VinVL-Paddle","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"paddle","reach":{"status":"unanswered"}},{"paper_slug":"vinvl-making-visual-representations-matter-in","repo_url":"https://github.com/microsoft/Oscar","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"vinvl-making-visual-representations-matter-in","repo_url":"https://github.com/mkhalil1998/EC601_Group_Project","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"unanswered"}},{"paper_slug":"vinvl-making-visual-representations-matter-in","repo_url":"https://github.com/yaolinli/capenrich","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null}],"tasks":[{"task_slug":"image-captioning","task_name":"Image Captioning"},{"task_slug":"image-text-matching","task_name":"Image-text matching"},{"task_slug":"object","task_name":"Object"},{"task_slug":"object-detection","task_name":"Object Detection"},{"task_slug":"visual-reasoning","task_name":"Visual Reasoning"},{"task_slug":"object-detection-1","task_name":"object-detection"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[{"leaderboard":"/sota/image-captioning-on-coco-captions","task":"Image Captioning","dataset":"COCO Captions","model":"VinVL","rank_in_archive_order":15,"of":41,"metrics":{"BLEU-4":"41.0","CIDER":"140.9","METEOR":"31.1","SPICE":"25.2"},"uses_additional_data":false},{"leaderboard":"/sota/image-captioning-on-nocaps-entire","task":"Image Captioning","dataset":"nocaps entire","model":"VinVL (Microsoft Cognitive Services + MSR)","rank_in_archive_order":13,"of":39,"metrics":{"B1":"81.59","B2":"65.15","B3":"45.04","B4":"26.15","CIDEr":"92.46","METEOR":"27.57","ROUGE-L":"56.96","SPICE":"13.07"},"uses_additional_data":false},{"leaderboard":"/sota/image-captioning-on-nocaps-in-domain","task":"Image Captioning","dataset":"nocaps in-domain","model":"VinVL (Microsoft Cognitive Services + MSR)","rank_in_archive_order":15,"of":41,"metrics":{"B1":"83.24","B2":"68.04","B3":"49.68","B4":"30.62","CIDEr":"97.99","METEOR":"29.51","ROUGE-L":"58.54","SPICE":"13.63"},"uses_additional_data":false},{"leaderboard":"/sota/image-captioning-on-nocaps-near-domain","task":"Image Captioning","dataset":"nocaps near-domain","model":"VinVL (Microsoft Cognitive Services + MSR)","rank_in_archive_order":13,"of":40,"metrics":{"B1":"82.77","B2":"66.94","B3":"47.02","B4":"27.97","CIDEr":"95.16","METEOR":"28.24","ROUGE-L":"57.95","SPICE":"13.36"},"uses_additional_data":false},{"leaderboard":"/sota/image-captioning-on-nocaps-out-of-domain","task":"Image Captioning","dataset":"nocaps out-of-domain","model":"VinVL (Microsoft Cognitive Services + MSR)","rank_in_archive_order":15,"of":40,"metrics":{"B1":"75.78","B2":"56.1","B3":"34.02","B4":"15.86","CIDEr":"78.01","METEOR":"23.55","ROUGE-L":"51.99","SPICE":"11.48"},"uses_additional_data":false},{"leaderboard":"/sota/image-captioning-on-nocaps-val-in-domain","task":"Image Captioning","dataset":"nocaps-val-in-domain","model":"VinVL","rank_in_archive_order":10,"of":11,"metrics":{"CIDEr":"103.1","Pre-train (#images)":"5.7M","SPICE":" 14.2"},"uses_additional_data":false},{"leaderboard":"/sota/image-captioning-on-nocaps-val-near-domain","task":"Image Captioning","dataset":"nocaps-val-near-domain","model":"VinVL","rank_in_archive_order":9,"of":10,"metrics":{"CIDEr":"96.1","Pre-train (#images)":"5.7M","SPICE":"13.8"},"uses_additional_data":false},{"leaderboard":"/sota/image-captioning-on-nocaps-val-out-domain","task":"Image Captioning","dataset":"nocaps-val-out-domain","model":"VinVL","rank_in_archive_order":10,"of":10,"metrics":{"CIDEr":" 88.3","Pretrain (#images)":"5.7M","SPICE":" 12.1"},"uses_additional_data":false},{"leaderboard":"/sota/image-captioning-on-nocaps-val-overall","task":"Image Captioning","dataset":"nocaps-val-overall","model":"VinVL","rank_in_archive_order":9,"of":11,"metrics":{"CIDEr":" 95.5","Pretrain (#images)":"5.7M","SPICE":" 13.5 "},"uses_additional_data":false},{"leaderboard":"/sota/image-text-matching-on-commercialadsdataset","task":"Image-text matching","dataset":"CommercialAdsDataset","model":"VinVL","rank_in_archive_order":2,"of":8,"metrics":{"ADD(S) AUC":"88.56"},"uses_additional_data":false},{"leaderboard":"/sota/visual-question-answering-on-gqa-test2019","task":"Visual Question Answering (VQA)","dataset":"GQA Test2019","model":"Single Model","rank_in_archive_order":11,"of":127,"metrics":{"Accuracy":"64.65","Binary":"82.63","Consistency":"94.35","Distribution":"4.72","Open":"48.77","Plausibility":"84.98","Validity":"96.62"},"uses_additional_data":false},{"leaderboard":"/sota/visual-question-answering-on-vqa-v2-test-std","task":"Visual Question Answering (VQA)","dataset":"VQA v2 test-std","model":"MSR + MS Cog. Svcs., X10 models","rank_in_archive_order":12,"of":38,"metrics":{"number":"62.55","other":"67.87","overall":"77.45","yes/no":"92.38"},"uses_additional_data":false},{"leaderboard":"/sota/visual-question-answering-on-vqa-v2-test-std","task":"Visual Question Answering (VQA)","dataset":"VQA v2 test-std","model":"MSR + MS Cog. Svcs.","rank_in_archive_order":13,"of":38,"metrics":{"number":"61.5","other":"66.68","overall":"76.63","yes/no":"92.04"},"uses_additional_data":false}],"syntology":{"syntology_url":"https://syntology.ai/paper/2101.00529","atlas_url":"https://app.syntology.ai/?focus=2101.00529","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2101.00529"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-25T09:33:49+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/microsoft/Oscar","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/cattidea/VinVL-Paddle","reach":{"status":"unanswered"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/mkhalil1998/EC601_Group_Project","reach":{"status":"unanswered"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/pzzhang/VinVL","reach":{"status":"unanswered"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/JoshuaPlacidi/MS-COCO-Object-Tags","reach":{"status":"unanswered"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/yaolinli/capenrich","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/JoshuaPlacidi/MS-COCO-Tags","reach":{"status":"unanswered"}}],"summary":{"ran":1,"ran_draft_wrong":1},"by_repo_kind":{"listed":{"samples":1,"ran":1,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":1,"samples":[{"code_sha256_prefix":"e17f1049b8148f71","entry":"make_data_sampler","repo":null,"repo_kind":null,"path":null,"file_url":null,"link_basis":"identical_code_first_harvested_elsewhere","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":null,"inline_ok":false,"mcp_get_code":{"code_sha256":"e17f1049b8148f71"}},{"code_sha256_prefix":"ad136de5b99f654e","entry":"process","repo":"yaolinli/capenrich","repo_kind":"listed","path":"oscar/end_uni_predict.py","file_url":"https://github.com/yaolinli/capenrich/blob/HEAD/oscar/end_uni_predict.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"ad136de5b99f654e"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}