{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/visual-spatial-reasoning","title":"Visual Spatial Reasoning","arxiv_id":"2205.00363","date":"2022-04-30","proceeding":null,"authors":["Fangyu Liu","Guy Emerson","Nigel Collier"],"abstract":"Spatial relations are a basic part of human cognition. However, they are expressed in natural language in a variety of ways, and previous work has suggested that current vision-and-language models (VLMs) struggle to capture relational information. In this paper, we present Visual Spatial Reasoning (VSR), a dataset containing more than 10k natural text-image pairs with 66 types of spatial relations in English (such as: under, in front of, and facing). While using a seemingly simple annotation format, we show how the dataset includes challenging linguistic phenomena, such as varying reference frames. We demonstrate a large gap between human and model performance: the human ceiling is above 95%, while state-of-the-art models only achieve around 70%. We observe that VLMs' by-relation performances have little correlation with the number of training examples and the tested models are in general incapable of recognising relations concerning the orientations of objects.","url_abs":"https://arxiv.org/abs/2205.00363v3","url_pdf":"https://arxiv.org/pdf/2205.00363v3.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"visual-spatial-reasoning","repo_url":"https://github.com/cambridgeltl/visual-spatial-reasoning","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"Apache-2.0"}},{"paper_slug":"visual-spatial-reasoning","repo_url":"https://github.com/sohojoe/clip_visual-spatial-reasoning","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":0,"framework":"pytorch","reach":{"status":"ok","spdx":"Apache-2.0"}},{"paper_slug":"visual-spatial-reasoning","repo_url":"https://github.com/ziyan-xiaoyu/spatialmqa","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"Apache-2.0"}},{"paper_slug":"visual-spatial-reasoning","repo_url":"https://github.com/MindCode-4/code-1/tree/main/vilt","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"mindspore","reach":null}],"tasks":[{"task_slug":"spatial-reasoning","task_name":"Spatial Reasoning"},{"task_slug":"visual-entailment","task_name":"Visual Entailment"},{"task_slug":"visual-reasoning","task_name":"Visual Reasoning"}],"methods":[{"method_slug":"clip","method_name":"CLIP"},{"method_slug":"lxmert","method_name":"LXMERT"},{"method_slug":"vilt","method_name":"ViLT"},{"method_slug":"visualbert","method_name":"VisualBERT"}],"datasets_introduced":[{"slug":"vsr","name":"VSR","full_name":"Visual Spatial Reasoning"}],"methods_introduced":[],"results":[{"leaderboard":"/sota/visual-reasoning-on-vsr","task":"Visual Reasoning","dataset":"VSR","model":"LXMERT","rank_in_archive_order":1,"of":5,"metrics":{"accuracy":"70.1"},"uses_additional_data":false},{"leaderboard":"/sota/visual-reasoning-on-vsr","task":"Visual Reasoning","dataset":"VSR","model":"ViLT","rank_in_archive_order":2,"of":5,"metrics":{"accuracy":"69.3"},"uses_additional_data":false},{"leaderboard":"/sota/visual-reasoning-on-vsr","task":"Visual Reasoning","dataset":"VSR","model":"CLIP (finetuned)","rank_in_archive_order":3,"of":5,"metrics":{"accuracy":"65.1"},"uses_additional_data":false},{"leaderboard":"/sota/visual-reasoning-on-vsr","task":"Visual Reasoning","dataset":"VSR","model":"CLIP (frozen)","rank_in_archive_order":4,"of":5,"metrics":{"accuracy":"56.0"},"uses_additional_data":false},{"leaderboard":"/sota/visual-reasoning-on-vsr","task":"Visual Reasoning","dataset":"VSR","model":"VisualBERT","rank_in_archive_order":5,"of":5,"metrics":{"accuracy":"55.2"},"uses_additional_data":false}],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=2205.00363","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.00363"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/cambridgeltl/visual-spatial-reasoning","reach":{"status":"ok","spdx":"Apache-2.0"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/MindCode-4/code-1/tree/main/vilt","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/ziyan-xiaoyu/spatialmqa","reach":{"status":"ok","spdx":"Apache-2.0"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/sohojoe/clip_visual-spatial-reasoning","reach":{"status":"ok","spdx":"Apache-2.0"}}],"summary":{"ran":5,"unverified":5},"by_repo_kind":{"official":{"samples":10,"ran":5,"repositories":2}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":0,"samples":[{"code_sha256_prefix":"e21bb85b8e1b4836","entry":"do_nms","repo":"cambridgeltl/visual-spatial-reasoning","repo_kind":"official","path":"feature_extraction/lxmert/modeling_frcnn.py","file_url":"https://github.com/cambridgeltl/visual-spatial-reasoning/blob/HEAD/feature_extraction/lxmert/modeling_frcnn.py","link_basis":"harvester_set","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"e21bb85b8e1b4836"}},{"code_sha256_prefix":"899d1e2736236351","entry":"is_remote_url","repo":"cambridgeltl/visual-spatial-reasoning","repo_kind":"official","path":"feature_extraction/lxmert/utils.py","file_url":"https://github.com/cambridgeltl/visual-spatial-reasoning/blob/HEAD/feature_extraction/lxmert/utils.py","link_basis":"harvester_set","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"899d1e2736236351"}},{"code_sha256_prefix":"8a78a755bda274a5","entry":"load_labels","repo":"cambridgeltl/visual-spatial-reasoning","repo_kind":"official","path":"feature_extraction/lxmert/utils.py","file_url":"https://github.com/cambridgeltl/visual-spatial-reasoning/blob/HEAD/feature_extraction/lxmert/utils.py","link_basis":"harvester_set","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"8a78a755bda274a5"}},{"code_sha256_prefix":"54a2fff601881b32","entry":"pad_list_tensors","repo":"cambridgeltl/visual-spatial-reasoning","repo_kind":"official","path":"feature_extraction/lxmert/modeling_frcnn.py","file_url":"https://github.com/cambridgeltl/visual-spatial-reasoning/blob/HEAD/feature_extraction/lxmert/modeling_frcnn.py","link_basis":"harvester_set","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"54a2fff601881b32"}},{"code_sha256_prefix":"77586e7080d66922","entry":"tryload","repo":"cambridgeltl/visual-spatial-reasoning","repo_kind":"official","path":"feature_extraction/lxmert/extracting_data.py","file_url":"https://github.com/cambridgeltl/visual-spatial-reasoning/blob/HEAD/feature_extraction/lxmert/extracting_data.py","link_basis":"harvester_set","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"77586e7080d66922"}},{"code_sha256_prefix":"eff2ebc2698c82dd","entry":"evaluate","repo":"sohojoe/clip_visual-spatial-reasoning","repo_kind":"official","path":"src/eval000.py","file_url":"https://github.com/sohojoe/clip_visual-spatial-reasoning/blob/HEAD/src/eval000.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"eff2ebc2698c82dd"}},{"code_sha256_prefix":"70d7553e03069ad3","entry":"evaluate","repo":"sohojoe/clip_visual-spatial-reasoning","repo_kind":"official","path":"src/eval001.py","file_url":"https://github.com/sohojoe/clip_visual-spatial-reasoning/blob/HEAD/src/eval001.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"70d7553e03069ad3"}},{"code_sha256_prefix":"7bdaa688884a132e","entry":"extract_features","repo":"cambridgeltl/visual-spatial-reasoning","repo_kind":"official","path":"feature_extraction/visualbert/extract_img_features.py","file_url":"https://github.com/cambridgeltl/visual-spatial-reasoning/blob/HEAD/feature_extraction/visualbert/extract_img_features.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"7bdaa688884a132e"}},{"code_sha256_prefix":"bd3a3f4bd8750da2","entry":"load_checkpoint","repo":"cambridgeltl/visual-spatial-reasoning","repo_kind":"official","path":"feature_extraction/lxmert/utils.py","file_url":"https://github.com/cambridgeltl/visual-spatial-reasoning/blob/HEAD/feature_extraction/lxmert/utils.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"bd3a3f4bd8750da2"}},{"code_sha256_prefix":"ea0acbc5664a2925","entry":"norm_box","repo":"cambridgeltl/visual-spatial-reasoning","repo_kind":"official","path":"feature_extraction/lxmert/modeling_frcnn.py","file_url":"https://github.com/cambridgeltl/visual-spatial-reasoning/blob/HEAD/feature_extraction/lxmert/modeling_frcnn.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"ea0acbc5664a2925"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}