{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/clipvisiontower","entry":"CLIPVisionTower","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":11,"n_papers_ran":3,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":11,"n_samples_ran":3,"n_samples_fingerprinted":2,"n_places":11,"n_places_pointer_only":3,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":0,"ran":3,"unverified":8},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2603.12930","paper":"/paper/arxiv-2603-12930","title":"Rethinking VLMs for Image Forgery Detection and Localization","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"sha0fengGuo/IFDL-VLM","path":"stage2/llava/model/language_model/llava_llama.py","file_url":"https://github.com/sha0fengGuo/IFDL-VLM/blob/HEAD/stage2/llava/model/language_model/llava_llama.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"04deb19e388bd0b0","mcp_get_code":{"code_sha256":"04deb19e388bd0b0"}},{"arxiv_id":"2505.19474","paper":"/paper/causal-llava-causal-disentanglement-for","title":"Causal-LLaVA: Causal Disentanglement for Mitigating Hallucination in Multimodal Large Language Models","date":"2025-05-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ignisavium/causal-llava","path":"confounders/get_projector_confounders.py","file_url":"https://github.com/ignisavium/causal-llava/blob/HEAD/confounders/get_projector_confounders.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"cedb1eb9694b29ff","mcp_get_code":{"code_sha256":"cedb1eb9694b29ff"}},{"arxiv_id":"2505.00275","paper":"/paper/adcare-vlm-leveraging-large-vision-language","title":"AdCare-VLM: Leveraging Large Vision Language Model (LVLM) to Monitor Long-Term Medication Adherence and Care","date":"2025-05-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"asad14053/AdCare-VLM","path":"videollava/model/language_model/llava_llama.py","file_url":"https://github.com/asad14053/AdCare-VLM/blob/HEAD/videollava/model/language_model/llava_llama.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6422a6162a5e6169","mcp_get_code":{"code_sha256":"6422a6162a5e6169"}},{"arxiv_id":"2503.24164","paper":null,"title":"arXiv:2503.24164","date":null,"month_inferred_from_arxiv_id":"2025-03","title_source":null,"repo":"vlm-svla/svla","path":"llava/model/language_model/llava_qwen.py","file_url":"https://github.com/vlm-svla/svla/blob/HEAD/llava/model/language_model/llava_qwen.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d808644e034e96d4","mcp_get_code":{"code_sha256":"d808644e034e96d4"}},{"arxiv_id":"2409.15657","paper":"/paper/mmpt-multimodal-prompt-tuning-for-zero-shot","title":"M$^2$PT: Multimodal Prompt Tuning for Zero-shot Instruction Learning","date":"2024-09-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"william-wang618/m2pt","path":"M2PT/model/llava_archPT.py","file_url":"https://github.com/william-wang618/m2pt/blob/HEAD/M2PT/model/llava_archPT.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f08131025c9de6a1","mcp_get_code":{"code_sha256":"f08131025c9de6a1"}},{"arxiv_id":"2406.19973","paper":"/paper/stllava-med-self-training-large-language-and","title":"STLLaVA-Med: Self-Training Large Language and Vision Assistant for Medical Question-Answering","date":"2024-06-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"heliossun/STLLaVA-Med","path":"llava/model/language_model/llava_llama.py","file_url":"https://github.com/heliossun/STLLaVA-Med/blob/HEAD/llava/model/language_model/llava_llama.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ee3792b1ae45cbb6","mcp_get_code":{"code_sha256":"ee3792b1ae45cbb6"}},{"arxiv_id":"2406.12246","paper":"/paper/trol-traversal-of-layers-for-large-language","title":"TroL: Traversal of Layers for Large Language and Vision Models","date":"2024-06-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ByungKwanLee/TroL","path":"trol/arch_internlm2/modeling_trol.py","file_url":"https://github.com/ByungKwanLee/TroL/blob/HEAD/trol/arch_internlm2/modeling_trol.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"10e2214bdfa59401","mcp_get_code":{"code_sha256":"10e2214bdfa59401"}},{"arxiv_id":"2403.04652","paper":"/paper/yi-open-foundation-models-by-01-ai","title":"Yi: Open Foundation Models by 01.AI","date":"2024-03-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"01-ai/yi","path":"VL/llava/model/llava_llama.py","file_url":"https://github.com/01-ai/yi/blob/HEAD/VL/llava/model/llava_llama.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f324cbfd797797c5","mcp_get_code":{"code_sha256":"f324cbfd797797c5"}},{"arxiv_id":"2402.18695","paper":"/paper/grounding-language-models-for-visual-entity","title":"Grounding Language Models for Visual Entity Recognition","date":"2024-02-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"MrZilinXiao/AutoVER","path":"LLaVA/llava/model/language_model/llava_llama.py","file_url":"https://github.com/MrZilinXiao/AutoVER/blob/HEAD/LLaVA/llava/model/language_model/llava_llama.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"c41616ca93fd79a8","mcp_get_code":{"code_sha256":"c41616ca93fd79a8"}},{"arxiv_id":"2306.02858","paper":"/paper/video-llama-an-instruction-tuned-audio-visual","title":"Video-LLaMA: An Instruction-tuned Audio-Visual Language Model for Video Understanding","date":"2023-06-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"damo-nlp-sg/videollama2","path":"videollama2/model/videollama2_arch.py","file_url":"https://github.com/damo-nlp-sg/videollama2/blob/HEAD/videollama2/model/videollama2_arch.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"351648024debfdef","mcp_get_code":{"code_sha256":"351648024debfdef"}},{"arxiv_id":"2304.08485","paper":"/paper/visual-instruction-tuning-1","title":"Visual Instruction Tuning","date":"2023-04-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dinhvietcuong1996/icme25-inova","path":"llava/model/language_model/llava_llama.py","file_url":"https://github.com/dinhvietcuong1996/icme25-inova/blob/HEAD/llava/model/language_model/llava_llama.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"98e6fcd805f498d1","mcp_get_code":{"code_sha256":"98e6fcd805f498d1"}}]}