{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/clip-2","entry":"CLIP","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":25,"n_papers_ran":4,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":26,"n_samples_ran":4,"n_samples_fingerprinted":0,"n_places":26,"n_places_pointer_only":16,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":0,"ran":4,"unverified":22},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2606.13288","paper":"/paper/arxiv-2606-13288","title":"Cross-Modal Masked Compositional Concept Modeling for Enhancing Visio-Linguistic Compositionality","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"hiker-lw/MACCO","path":"src/open_clip_code/MACCO_variant/MACCO_text_image.py","file_url":"https://github.com/hiker-lw/MACCO/blob/HEAD/src/open_clip_code/MACCO_variant/MACCO_text_image.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c88c1b3bdd5f5cc8","mcp_get_code":{"code_sha256":"c88c1b3bdd5f5cc8"}},{"arxiv_id":"2510.20162","paper":"/paper/arxiv-2510-20162","title":"TOMCAT : Test-time Comprehensive Knowledge Accumulation for Compositional Zero-Shot Learning","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"xud-yan/TOMCAT","path":"model/tomcat_bm.py","file_url":"https://github.com/xud-yan/TOMCAT/blob/HEAD/model/tomcat_bm.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d43285ed12c61ef7","mcp_get_code":{"code_sha256":"d43285ed12c61ef7"}},{"arxiv_id":"2510.18583","paper":"/paper/arxiv-2510-18583","title":"CovMatch: Cross-Covariance Guided Multimodal Dataset Distillation with Trainable Text Encoder","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"Yongalls/CovMatch","path":"src/model.py","file_url":"https://github.com/Yongalls/CovMatch/blob/HEAD/src/model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"23942a1e71ee1ce4","mcp_get_code":{"code_sha256":"23942a1e71ee1ce4"}},{"arxiv_id":"2505.10289","paper":"/paper/msci-addressing-clip-s-inherent-limitations","title":"MSCI: Addressing CLIP's Inherent Limitations for Compositional Zero-Shot Learning","date":"2025-05-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ltpwy/MSCI","path":"MSCI/code/model/Mutifuse_new.py","file_url":"https://github.com/ltpwy/MSCI/blob/HEAD/MSCI/code/model/Mutifuse_new.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"255bc63ca03ae8d9","mcp_get_code":{"code_sha256":"255bc63ca03ae8d9"}},{"arxiv_id":"2504.12104","paper":"/paper/logits-deconfusion-with-clip-for-few-shot","title":"Logits DeConfusion with CLIP for Few-Shot Learning","date":"2025-04-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"LiShuo1001/LDC","path":"clip_ldc/model.py","file_url":"https://github.com/LiShuo1001/LDC/blob/HEAD/clip_ldc/model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"65fe97557e0cb835","mcp_get_code":{"code_sha256":"65fe97557e0cb835"}},{"arxiv_id":"2502.03549","paper":"/paper/kronecker-mask-and-interpretive-prompts-are","title":"Kronecker Mask and Interpretive Prompts are Language-Action Video Learners","date":"2025-02-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yjyddq/CLAVER","path":"models/claver.py","file_url":"https://github.com/yjyddq/CLAVER/blob/HEAD/models/claver.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ea3373815e1a7a63","mcp_get_code":{"code_sha256":"ea3373815e1a7a63"}},{"arxiv_id":"2501.13859","paper":"/paper/dual-modal-prototype-joint-learning-for","title":"Dual-Modal Prototype Joint Learning for Compositional Zero-Shot Learning","date":"2025-01-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"codefish12-09/VP_CMJL","path":"model/VP_CMJL.py","file_url":"https://github.com/codefish12-09/VP_CMJL/blob/HEAD/model/VP_CMJL.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"b952dc9e74cd31a0","mcp_get_code":{"code_sha256":"b952dc9e74cd31a0"}},{"arxiv_id":"2409.06809","paper":"/paper/detailclip-detail-oriented-clip-for-fine","title":"DetailCLIP: Detail-Oriented CLIP for Fine-Grained Tasks","date":"2024-09-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"KishoreP1/DetailCLIP","path":"models.py","file_url":"https://github.com/KishoreP1/DetailCLIP/blob/HEAD/models.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"8d7b5685c27c8566","mcp_get_code":{"code_sha256":"8d7b5685c27c8566"}},{"arxiv_id":"2407.15795","paper":"/paper/adaclip-adapting-clip-with-hybrid-learnable","title":"AdaCLIP: Adapting CLIP with Hybrid Learnable Prompts for Zero-Shot Anomaly Detection","date":"2024-07-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"caoyunkang/AdaCLIP","path":"method/adaclip.py","file_url":"https://github.com/caoyunkang/AdaCLIP/blob/HEAD/method/adaclip.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"7dd2e9759231f093","mcp_get_code":{"code_sha256":"7dd2e9759231f093"}},{"arxiv_id":"2404.19228","paper":"/paper/understanding-multimodal-contrastive-learning-1","title":"Weighted Point Cloud Embedding for Multimodal Contrastive Learning Toward Optimal Similarity Metric","date":"2024-04-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sony/wpse","path":"models.py","file_url":"https://github.com/sony/wpse/blob/HEAD/models.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"07ea912271e29c92","mcp_get_code":{"code_sha256":"07ea912271e29c92"}},{"arxiv_id":"2403.16005","paper":"/paper/knowledge-enhanced-dual-stream-zero-shot","title":"Knowledge-Enhanced Dual-stream Zero-shot Composed Image Retrieval","date":"2024-03-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"suoych/KEDs","path":"src/model/model.py","file_url":"https://github.com/suoych/KEDs/blob/HEAD/src/model/model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3949802de58e42db","mcp_get_code":{"code_sha256":"3949802de58e42db"}},{"arxiv_id":"2309.12867","paper":"/paper/accurate-and-fast-compressed-video-captioning","title":"Accurate and Fast Compressed Video Captioning","date":"2023-09-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"acherstyx/CoCap","path":"cocap/modules/compressed_video/compressed_video_transformer.py","file_url":"https://github.com/acherstyx/CoCap/blob/HEAD/cocap/modules/compressed_video/compressed_video_transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f17aef0f80b06b5e","mcp_get_code":{"code_sha256":"f17aef0f80b06b5e"}},{"arxiv_id":"2309.01083","paper":"/paper/chinese-text-recognition-with-a-pre-trained","title":"Chinese Text Recognition with A Pre-Trained CLIP-Like Model Through Image-IDS Aligning","date":"2023-09-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"FudanVI/FudanOCR","path":"image-ids-CTR/CCR-CLIP/model.py","file_url":"https://github.com/FudanVI/FudanOCR/blob/HEAD/image-ids-CTR/CCR-CLIP/model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"fe215b87ac8d71bf","mcp_get_code":{"code_sha256":"fe215b87ac8d71bf"}},{"arxiv_id":"2306.09244","paper":"/paper/text-promptable-surgical-instrument","title":"Text Promptable Surgical Instrument Segmentation with Vision-Language Models","date":"2023-06-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"franciszzj/tp-sis","path":"model/segmenter.py","file_url":"https://github.com/franciszzj/tp-sis/blob/HEAD/model/segmenter.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"326e17096eb218fb","mcp_get_code":{"code_sha256":"326e17096eb218fb"}},{"arxiv_id":"2305.13500","paper":"/paper/learning-emotion-representations-from-verbal-1","title":"Learning Emotion Representations from Verbal and Nonverbal Communication","date":"2023-05-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Xeaver/EmotionCLIP","path":"src/models/base.py","file_url":"https://github.com/Xeaver/EmotionCLIP/blob/HEAD/src/models/base.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bd68a70f00670d88","mcp_get_code":{"code_sha256":"bd68a70f00670d88"}},{"arxiv_id":"2304.04231","paper":"/paper/crowdclip-unsupervised-crowd-counting-via","title":"CrowdCLIP: Unsupervised Crowd Counting via Vision-Language Model","date":"2023-04-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dk-liang/CrowdCLIP","path":"CLIP/clip/model.py","file_url":"https://github.com/dk-liang/CrowdCLIP/blob/HEAD/CLIP/clip/model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"2daf328eb961a325","mcp_get_code":{"code_sha256":"2daf328eb961a325"}},{"arxiv_id":"2303.15343","paper":"/paper/sigmoid-loss-for-language-image-pre-training","title":"Sigmoid Loss for Language Image Pre-Training","date":"2023-03-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ramanakshay/clip","path":"src/model/model.py","file_url":"https://github.com/ramanakshay/clip/blob/HEAD/src/model/model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"7f362a66c1752126","mcp_get_code":{"code_sha256":"7f362a66c1752126"}},{"arxiv_id":"2303.15343","paper":"/paper/sigmoid-loss-for-language-image-pre-training","title":"Sigmoid Loss for Language Image Pre-Training","date":"2023-03-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"apple/ml-mobileclip","path":"mobileclip/clip.py","file_url":"https://github.com/apple/ml-mobileclip/blob/HEAD/mobileclip/clip.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"a8fa6ab303249a21","mcp_get_code":{"code_sha256":"a8fa6ab303249a21"}},{"arxiv_id":"2303.14369","paper":"/paper/video-text-as-game-players-hierarchical","title":"Video-Text as Game Players: Hierarchical Banzhaf Interaction for Cross-Modal Representation Learning","date":"2023-03-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jpthu17/dicosa","path":"tvr/models/modeling.py","file_url":"https://github.com/jpthu17/dicosa/blob/HEAD/tvr/models/modeling.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"aacf9b08feb3ea59","mcp_get_code":{"code_sha256":"aacf9b08feb3ea59"}},{"arxiv_id":"2303.12501","paper":"/paper/cross-modal-implicit-relation-reasoning-and","title":"Cross-Modal Implicit Relation Reasoning and Aligning for Text-to-Image Person Retrieval","date":"2023-03-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"anosorae/irra","path":"model/build.py","file_url":"https://github.com/anosorae/irra/blob/HEAD/model/build.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"36986ffbd858f742","mcp_get_code":{"code_sha256":"36986ffbd858f742"}},{"arxiv_id":"2301.06958","paper":"/paper/masked-visual-reconstruction-in-language","title":"RILS: Masked Visual Reconstruction in Language Semantic Space","date":"2023-01-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hustvl/RILS","path":"models.py","file_url":"https://github.com/hustvl/RILS/blob/HEAD/models.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"b08d6847b3c9d093","mcp_get_code":{"code_sha256":"b08d6847b3c9d093"}},{"arxiv_id":"2205.14459","paper":"/paper/cyclip-cyclic-contrastive-language-image","title":"CyCLIP: Cyclic Contrastive Language-Image Pretraining","date":"2022-05-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"goel-shashank/CyCLIP","path":"pkgs/openai/model.py","file_url":"https://github.com/goel-shashank/CyCLIP/blob/HEAD/pkgs/openai/model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c85999b6f5cbfcc7","mcp_get_code":{"code_sha256":"c85999b6f5cbfcc7"}},{"arxiv_id":"2105.01879","paper":"/paper/mos-towards-scaling-out-of-distribution","title":"MOS: Towards Scaling Out-of-distribution Detection for Large Semantic Space","date":"2021-05-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ma-kjh/CMA-OoDD","path":"clip/model.py","file_url":"https://github.com/ma-kjh/CMA-OoDD/blob/HEAD/clip/model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ef2d645ece6b584c","mcp_get_code":{"code_sha256":"ef2d645ece6b584c"}},{"arxiv_id":"2102.05918","paper":"/paper/scaling-up-visual-and-vision-language","title":"Scaling Up Visual and Vision-Language Representation Learning With Noisy Text Supervision","date":"2021-02-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"facebookresearch/metaclip","path":"src/mini_clip/model.py","file_url":"https://github.com/facebookresearch/metaclip/blob/HEAD/src/mini_clip/model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"349072ec3924682a","mcp_get_code":{"code_sha256":"349072ec3924682a"}},{"arxiv_id":"2011.10566","paper":"/paper/exploring-simple-siamese-representation","title":"Exploring Simple Siamese Representation Learning","date":"2020-11-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"facebookresearch/clip-rocket","path":"models.py","file_url":"https://github.com/facebookresearch/clip-rocket/blob/HEAD/models.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"4d45d7d0856d7377","mcp_get_code":{"code_sha256":"4d45d7d0856d7377"}},{"arxiv_id":"2025.findings-emnlp.28","paper":null,"title":"arXiv:2025.findings-emnlp.28","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"BUAAPY/ProPy","path":"modules/clip_propy.py","file_url":"https://github.com/BUAAPY/ProPy/blob/HEAD/modules/clip_propy.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"7416db18e5cb12de","mcp_get_code":{"code_sha256":"7416db18e5cb12de"}}]}