{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/visualtransformer","entry":"VisualTransformer","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":12,"n_papers_ran":10,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":12,"n_samples_ran":10,"n_samples_fingerprinted":1,"n_places":12,"n_places_pointer_only":6,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":0,"ran":10,"unverified":2},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2606.13288","paper":"/paper/arxiv-2606-13288","title":"Cross-Modal Masked Compositional Concept Modeling for Enhancing Visio-Linguistic Compositionality","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"hiker-lw/MACCO","path":"src/open_clip_code/MACCO_variant/MACCO_text_image.py","file_url":"https://github.com/hiker-lw/MACCO/blob/HEAD/src/open_clip_code/MACCO_variant/MACCO_text_image.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a210faa2dd758b01","mcp_get_code":{"code_sha256":"a210faa2dd758b01"}},{"arxiv_id":"2510.20162","paper":"/paper/arxiv-2510-20162","title":"TOMCAT : Test-time Comprehensive Knowledge Accumulation for Compositional Zero-Shot Learning","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"xud-yan/TOMCAT","path":"model/tomcat_bm.py","file_url":"https://github.com/xud-yan/TOMCAT/blob/HEAD/model/tomcat_bm.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"88d8f3ddec06117f","mcp_get_code":{"code_sha256":"88d8f3ddec06117f"}},{"arxiv_id":"2505.10289","paper":"/paper/msci-addressing-clip-s-inherent-limitations","title":"MSCI: Addressing CLIP's Inherent Limitations for Compositional Zero-Shot Learning","date":"2025-05-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ltpwy/MSCI","path":"MSCI/code/model/Mutifuse_new.py","file_url":"https://github.com/ltpwy/MSCI/blob/HEAD/MSCI/code/model/Mutifuse_new.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"257a8b6ed7a22d7e","mcp_get_code":{"code_sha256":"257a8b6ed7a22d7e"}},{"arxiv_id":"2501.13859","paper":"/paper/dual-modal-prototype-joint-learning-for","title":"Dual-Modal Prototype Joint Learning for Compositional Zero-Shot Learning","date":"2025-01-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"codefish12-09/VP_CMJL","path":"model/VP_CMJL.py","file_url":"https://github.com/codefish12-09/VP_CMJL/blob/HEAD/model/VP_CMJL.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c14d889910ecfad4","mcp_get_code":{"code_sha256":"c14d889910ecfad4"}},{"arxiv_id":"2403.16005","paper":"/paper/knowledge-enhanced-dual-stream-zero-shot","title":"Knowledge-Enhanced Dual-stream Zero-shot Composed Image Retrieval","date":"2024-03-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"suoych/KEDs","path":"src/model/model.py","file_url":"https://github.com/suoych/KEDs/blob/HEAD/src/model/model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6eeb0850e867ee04","mcp_get_code":{"code_sha256":"6eeb0850e867ee04"}},{"arxiv_id":"2403.09500","paper":"/paper/faceptor-a-generalist-model-for-face","title":"Faceptor: A Generalist Model for Face Perception","date":"2024-03-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lxq1000/Faceptor","path":"faceptor_project/core/model/backbone/farl_vit.py","file_url":"https://github.com/lxq1000/Faceptor/blob/HEAD/faceptor_project/core/model/backbone/farl_vit.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e156ac655f0fc2d7","mcp_get_code":{"code_sha256":"e156ac655f0fc2d7"}},{"arxiv_id":"2308.08428","paper":"/paper/alip-adaptive-language-image-pre-training","title":"ALIP: Adaptive Language-Image Pre-training with Synthetic Caption","date":"2023-08-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"deepglint/alip","path":"src/open_alip/model.py","file_url":"https://github.com/deepglint/alip/blob/HEAD/src/open_alip/model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4a20fae4815b1a33","mcp_get_code":{"code_sha256":"4a20fae4815b1a33"}},{"arxiv_id":"2305.13500","paper":"/paper/learning-emotion-representations-from-verbal-1","title":"Learning Emotion Representations from Verbal and Nonverbal Communication","date":"2023-05-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Xeaver/EmotionCLIP","path":"src/models/base.py","file_url":"https://github.com/Xeaver/EmotionCLIP/blob/HEAD/src/models/base.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c857cf6246801938","mcp_get_code":{"code_sha256":"c857cf6246801938"}},{"arxiv_id":"2303.14369","paper":"/paper/video-text-as-game-players-hierarchical","title":"Video-Text as Game Players: Hierarchical Banzhaf Interaction for Cross-Modal Representation Learning","date":"2023-03-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jpthu17/dicosa","path":"tvr/models/modeling.py","file_url":"https://github.com/jpthu17/dicosa/blob/HEAD/tvr/models/modeling.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bab52ac82134c899","mcp_get_code":{"code_sha256":"bab52ac82134c899"}},{"arxiv_id":"2205.14459","paper":"/paper/cyclip-cyclic-contrastive-language-image","title":"CyCLIP: Cyclic Contrastive Language-Image Pretraining","date":"2022-05-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"goel-shashank/CyCLIP","path":"pkgs/openai/model.py","file_url":"https://github.com/goel-shashank/CyCLIP/blob/HEAD/pkgs/openai/model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"385ab82ec3447e70","mcp_get_code":{"code_sha256":"385ab82ec3447e70"}},{"arxiv_id":"2102.05918","paper":"/paper/scaling-up-visual-and-vision-language","title":"Scaling Up Visual and Vision-Language Representation Learning With Noisy Text Supervision","date":"2021-02-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"facebookresearch/metaclip","path":"src/mini_clip/model.py","file_url":"https://github.com/facebookresearch/metaclip/blob/HEAD/src/mini_clip/model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"29c96c042002f376","mcp_get_code":{"code_sha256":"29c96c042002f376"}},{"arxiv_id":"2025.findings-emnlp.28","paper":null,"title":"arXiv:2025.findings-emnlp.28","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"BUAAPY/ProPy","path":"modules/clip_propy.py","file_url":"https://github.com/BUAAPY/ProPy/blob/HEAD/modules/clip_propy.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"66f03e0c46196b72","mcp_get_code":{"code_sha256":"66f03e0c46196b72"}}]}