{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/expand2square","entry":"expand2square","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":78,"n_papers_ran":76,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":10,"n_samples_ran":7,"n_samples_fingerprinted":0,"n_places":80,"n_places_pointer_only":34,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":2,"ran_fixture":0,"ran":5,"unverified":3},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2605.27378","paper":"/paper/arxiv-2605-27378","title":"OralAgent: Integrating Reasoning, Tools, and Knowledge for Interactive Dental Image Analysis","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"isjinghao/OralAgent","path":"oralagent/llava/mm_utils.py","file_url":"https://github.com/isjinghao/OralAgent/blob/HEAD/oralagent/llava/mm_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"908505ed68ff4871","mcp_get_code":{"code_sha256":"908505ed68ff4871"}},{"arxiv_id":"2603.21426","paper":"/paper/arxiv-2603-21426","title":"Uncertainty-Aware Knowledge Distillation for Multimodal Large Language Models","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"Jingchensun/beta-kd","path":"mobilevlm/utils.py","file_url":"https://github.com/Jingchensun/beta-kd/blob/HEAD/mobilevlm/utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2602.09483","paper":"/paper/arxiv-2602-09483","title":"Beyond Next-Token Alignment: Distilling Multimodal Large Language Models via Token Interactions","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"lchen1019/Align-TI","path":"alignti/mm_utils.py","file_url":"https://github.com/lchen1019/Align-TI/blob/HEAD/alignti/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2601.17918","paper":"/paper/arxiv-2601-17918","title":"Benchmarking Direct Preference Optimization for Medical Large Vision-Language Models","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"dmis-lab/med-vlm-dpo","path":"inference/LLaVA-Med/llava/mm_utils.py","file_url":"https://github.com/dmis-lab/med-vlm-dpo/blob/HEAD/inference/LLaVA-Med/llava/mm_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"908505ed68ff4871","mcp_get_code":{"code_sha256":"908505ed68ff4871"}},{"arxiv_id":"2601.11522","paper":"/paper/arxiv-2601-11522","title":"UniX: Unifying Autoregression and Diffusion for Chest X-Ray Understanding and Generation","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"ZrH42/UniX","path":"modeling/unix_vlm/models/image_processing_vlm.py","file_url":"https://github.com/ZrH42/UniX/blob/HEAD/modeling/unix_vlm/models/image_processing_vlm.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2509.24365","paper":"/paper/arxiv-2509-24365","title":"Uni-X: Mitigating Modality Conflict with a Two-End-Separated Architecture for Unified Multimodal Models","date":null,"month_inferred_from_arxiv_id":"2025-09","title_source":"syntology","repo":"CURRENTF/Uni-X","path":"uni_arch/mm_utils.py","file_url":"https://github.com/CURRENTF/Uni-X/blob/HEAD/uni_arch/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2507.18300","paper":null,"title":"arXiv:2507.18300","date":null,"month_inferred_from_arxiv_id":"2025-07","title_source":null,"repo":"360CVGroup/LMM-Det","path":"llava/mm_utils.py","file_url":"https://github.com/360CVGroup/LMM-Det/blob/HEAD/llava/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2507.04976","paper":null,"title":"arXiv:2507.04976","date":null,"month_inferred_from_arxiv_id":"2025-07","title_source":null,"repo":"EsYoon7/UVQA","path":"llava/mm_utils.py","file_url":"https://github.com/EsYoon7/UVQA/blob/HEAD/llava/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2505.24875","paper":"/paper/reasongen-r1-cot-for-autoregressive-image","title":"ReasonGen-R1: CoT for Autoregressive Image generation models through SFT and RL","date":"2025-05-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Franklin-Zhang0/ReasonGen-R1","path":"Janus/janus/models/image_processing_vlm.py","file_url":"https://github.com/Franklin-Zhang0/ReasonGen-R1/blob/HEAD/Janus/janus/models/image_processing_vlm.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2505.20753","paper":"/paper/understand-think-and-answer-advancing-visual","title":"Understand, Think, and Answer: Advancing Visual Reasoning with Large Multimodal Models","date":"2025-05-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jefferyzhan/griffon","path":"griffon/mm_utils.py","file_url":"https://github.com/jefferyzhan/griffon/blob/HEAD/griffon/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2502.06788","paper":"/paper/evev2-improved-baselines-for-encoder-free","title":"EVEv2: Improved Baselines for Encoder-Free Vision-Language Models","date":"2025-02-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"baaivision/EVE","path":"EVEv1/eve/mm_utils.py","file_url":"https://github.com/baaivision/EVE/blob/HEAD/EVEv1/eve/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2502.02673","paper":"/paper/medrax-medical-reasoning-agent-for-chest-x","title":"MedRAX: Medical Reasoning Agent for Chest X-ray","date":"2025-02-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bowang-lab/medrax","path":"medrax/llava/mm_utils.py","file_url":"https://github.com/bowang-lab/medrax/blob/HEAD/medrax/llava/mm_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"908505ed68ff4871","mcp_get_code":{"code_sha256":"908505ed68ff4871"}},{"arxiv_id":"2501.13106","paper":"/paper/videollama-3-frontier-multimodal-foundation","title":"VideoLLaMA 3: Frontier Multimodal Foundation Models for Image and Video Understanding","date":"2025-01-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"damo-nlp-sg/videollama3","path":"videollama3/mm_utils.py","file_url":"https://github.com/damo-nlp-sg/videollama3/blob/HEAD/videollama3/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2412.14006","paper":"/paper/instructseg-unifying-instructed-visual","title":"InstructSeg: Unifying Instructed Visual Segmentation with Multi-modal Large Language Models","date":"2024-12-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"congvvc/instructseg","path":"instructseg/model/mipha/mm_utils.py","file_url":"https://github.com/congvvc/instructseg/blob/HEAD/instructseg/model/mipha/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2412.09951","paper":"/paper/wisead-knowledge-augmented-end-to-end","title":"WiseAD: Knowledge Augmented End-to-End Autonomous Driving with Vision-Language Model","date":"2024-12-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wyddmw/WiseAD","path":"mobilevlm/utils.py","file_url":"https://github.com/wyddmw/WiseAD/blob/HEAD/mobilevlm/utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2412.04332","paper":"/paper/liquid-language-models-are-scalable-multi","title":"Liquid: Language Models are Scalable Multi-modal Generators","date":"2024-12-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"foundationvision/liquid","path":"evaluation/inference_i2t.py","file_url":"https://github.com/foundationvision/liquid/blob/HEAD/evaluation/inference_i2t.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2412.04317","paper":"/paper/flashsloth-lightning-multimodal-large","title":"FlashSloth: Lightning Multimodal Large Language Models via Embedded Visual Compression","date":"2024-12-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"codefanw/flashsloth","path":"flashsloth/mm_utils.py","file_url":"https://github.com/codefanw/flashsloth/blob/HEAD/flashsloth/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2411.17606","paper":"/paper/hyperseg-towards-universal-visual","title":"HyperSeg: Towards Universal Visual Segmentation with Large Language Model","date":"2024-11-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"congvvc/HyperSeg","path":"hyperseg/model/mipha/mm_utils.py","file_url":"https://github.com/congvvc/HyperSeg/blob/HEAD/hyperseg/model/mipha/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2411.16724","paper":"/paper/devils-in-middle-layers-of-large-vision","title":"Devils in Middle Layers of Large Vision-Language Models: Interpreting, Detecting and Mitigating Object Hallucinations via Attention Lens","date":"2024-11-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zhangqijiang07/middle_layers_indicating_hallucinations","path":"eval_data_loader.py","file_url":"https://github.com/zhangqijiang07/middle_layers_indicating_hallucinations/blob/HEAD/eval_data_loader.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2411.02712","paper":"/paper/v-dpo-mitigating-hallucination-in-large","title":"V-DPO: Mitigating Hallucination in Large Vision Language Models via Vision-Guided Direct Preference Optimization","date":"2024-11-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yuxixie/v-dpo","path":"llava_dpo/mm_utils.py","file_url":"https://github.com/yuxixie/v-dpo/blob/HEAD/llava_dpo/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2410.04780","paper":"/paper/mitigating-modality-prior-induced","title":"Mitigating Modality Prior-Induced Hallucinations in Multimodal Large Language Models via Deciphering Attention Causality","date":"2024-10-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"The-Martyr/CausalMM","path":"llava-1.5/experiments/llava/mm_utils.py","file_url":"https://github.com/The-Martyr/CausalMM/blob/HEAD/llava-1.5/experiments/llava/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2409.17143","paper":"/paper/attention-prompting-on-image-for-large-vision","title":"Attention Prompting on Image for Large Vision-Language Models","date":"2024-09-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2409.16261","paper":"/paper/cdchat-a-large-multimodal-model-for-remote","title":"CDChat: A Large Multimodal Model for Remote Sensing Change Description","date":"2024-09-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"techmn/cdchat","path":"cdchat/mm_utils.py","file_url":"https://github.com/techmn/cdchat/blob/HEAD/cdchat/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2409.10197","paper":"/paper/fit-and-prune-fast-and-training-free-visual","title":"Fit and Prune: Fast and Training-free Visual Token Pruning for Multi-modal Large Language Models","date":"2024-09-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ywh187/fitprune","path":"LLaVA_1.5/llava/mm_utils.py","file_url":"https://github.com/ywh187/fitprune/blob/HEAD/LLaVA_1.5/llava/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2409.08582","paper":"/paper/changechat-an-interactive-model-for-remote","title":"ChangeChat: An Interactive Model for Remote Sensing Change Analysis via Multimodal Instruction Tuning","date":"2024-09-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hanlinwu/changechat","path":"changechat/mm_utils.py","file_url":"https://github.com/hanlinwu/changechat/blob/HEAD/changechat/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2409.01179","paper":"/paper/recoverable-compression-a-multimodal-vision","title":"Recoverable Compression: A Multimodal Vision Token Recovery Mechanism Guided by Text Information","date":"2024-09-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"banjiuyufen/Recoverable-Compression","path":"llava/mm_utils.py","file_url":"https://github.com/banjiuyufen/Recoverable-Compression/blob/HEAD/llava/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2408.12902","paper":"/paper/iaa-inner-adaptor-architecture-empowers","title":"IAA: Inner-Adaptor Architecture Empowers Frozen Large Language Model with Multimodal Capabilities","date":"2024-08-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"360cvgroup/inner-adaptor-architecture","path":"iaa/mm_utils.py","file_url":"https://github.com/360cvgroup/inner-adaptor-architecture/blob/HEAD/iaa/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2408.08640","paper":"/paper/math-puma-progressive-upward-multimodal","title":"Math-PUMA: Progressive Upward Multimodal Alignment to Enhance Mathematical Reasoning","date":"2024-08-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wwzhuang01/math-puma","path":"models/deepseek_math/image_processing_vlm.py","file_url":"https://github.com/wwzhuang01/math-puma/blob/HEAD/models/deepseek_math/image_processing_vlm.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"GPL-3.0","inline_ok":false,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2407.21771","paper":"/paper/paying-more-attention-to-image-a-training","title":"Paying More Attention to Image: A Training-Free Method for Alleviating Hallucination in LVLMs","date":"2024-07-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hasanar1f/llava-hallunication-fix","path":"modPAI/llava/mm_utils.py","file_url":"https://github.com/hasanar1f/llava-hallunication-fix/blob/HEAD/modPAI/llava/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2407.08303","paper":"/paper/densefusion-1m-merging-vision-experts-for","title":"DenseFusion-1M: Merging Vision Experts for Comprehensive Multimodal Perception","date":"2024-07-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"baaivision/DenseFusion","path":"densefusion/mm_utils.py","file_url":"https://github.com/baaivision/DenseFusion/blob/HEAD/densefusion/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2406.20098","paper":"/paper/web2code-a-large-scale-webpage-to-code","title":"Web2Code: A Large-scale Webpage-to-Code Dataset and Evaluation Framework for Multimodal LLMs","date":"2024-06-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"MBZUAI-LLM/web2code","path":"web2code/llava/mm_utils.py","file_url":"https://github.com/MBZUAI-LLM/web2code/blob/HEAD/web2code/llava/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2406.13173","paper":"/paper/biomedical-visual-instruction-tuning-with","title":"Biomedical Visual Instruction Tuning with Clinician Preference Alignment","date":"2024-06-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mao1207/BioMed-VITAL","path":"backbone/mm_utils.py","file_url":"https://github.com/mao1207/BioMed-VITAL/blob/HEAD/backbone/mm_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"908505ed68ff4871","mcp_get_code":{"code_sha256":"908505ed68ff4871"}},{"arxiv_id":"2406.12235","paper":"/paper/holmes-vad-towards-unbiased-and-explainable","title":"Holmes-VAD: Towards Unbiased and Explainable Video Anomaly Detection via Multi-modal LLM","date":"2024-06-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pipixin321/holmesvad","path":"videollava/mm_utils.py","file_url":"https://github.com/pipixin321/holmesvad/blob/HEAD/videollava/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2406.11839","paper":"/paper/mdpo-conditional-preference-optimization-for","title":"mDPO: Conditional Preference Optimization for Multimodal Large Language Models","date":"2024-06-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"luka-group/mDPO","path":"bunny/bunny_utils/util/mm_utils.py","file_url":"https://github.com/luka-group/mDPO/blob/HEAD/bunny/bunny_utils/util/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2406.10215","paper":"/paper/devbench-a-multimodal-developmental-benchmark","title":"DevBench: A multimodal developmental benchmark for language learning","date":"2024-06-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"alvinwmtan/dev-bench","path":"model_classes/modeling_tinyllava_phi.py","file_url":"https://github.com/alvinwmtan/dev-bench/blob/HEAD/model_classes/modeling_tinyllava_phi.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2406.08100","paper":"/paper/multimodal-table-understanding","title":"Multimodal Table Understanding","date":"2024-06-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"spursgozmy/table-llava","path":"llava/mm_utils.py","file_url":"https://github.com/spursgozmy/table-llava/blob/HEAD/llava/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2406.02884","paper":"/paper/posterllava-constructing-a-unified-multi","title":"PosterLLaVa: Constructing a Unified Multi-modal Layout Generator with LLM","date":"2024-06-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"posterllava/posterllava","path":"llava/mm_utils.py","file_url":"https://github.com/posterllava/posterllava/blob/HEAD/llava/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2405.19298","paper":"/paper/adaptive-image-quality-assessment-via","title":"Adaptive Image Quality Assessment via Teaching Large Multimodal Model to Compare","date":"2024-05-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Q-Future/Compare2Score","path":"q_align/mm_utils.py","file_url":"https://github.com/Q-Future/Compare2Score/blob/HEAD/q_align/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2405.19298","paper":"/paper/adaptive-image-quality-assessment-via","title":"Adaptive Image Quality Assessment via Teaching Large Multimodal Model to Compare","date":"2024-05-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Q-Future/Compare2Score","path":"q_align/model/modeling_mplug_owl2.py","file_url":"https://github.com/Q-Future/Compare2Score/blob/HEAD/q_align/model/modeling_mplug_owl2.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0d475d995befa10b","mcp_get_code":{"code_sha256":"0d475d995befa10b"}},{"arxiv_id":"2405.15738","paper":"/paper/convllava-hierarchical-backbones-as-visual","title":"ConvLLaVA: Hierarchical Backbones as Visual Encoder for Large Multimodal Models","date":"2024-05-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"alibaba/conv-llava","path":"llava/eval/evaluate_grounding.py","file_url":"https://github.com/alibaba/conv-llava/blob/HEAD/llava/eval/evaluate_grounding.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2405.14974","paper":"/paper/lova3-learning-to-visual-question-answering","title":"LOVA3: Learning to Visual Question Answering, Asking and Assessment","date":"2024-05-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"showlab/LOVA3","path":"llava/mm_utils.py","file_url":"https://github.com/showlab/LOVA3/blob/HEAD/llava/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2405.12107","paper":"/paper/imp-highly-capable-large-multimodal-models","title":"Imp: Highly Capable Large Multimodal Models for Mobile Devices","date":"2024-05-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"milvlg/imp","path":"imp_llava/mm_utils.py","file_url":"https://github.com/milvlg/imp/blob/HEAD/imp_llava/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2405.11165","paper":"/paper/automated-multi-level-preference-for-mllms","title":"Automated Multi-level Preference for MLLMs","date":"2024-05-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"takomc/amp","path":"llava/mm_utils.py","file_url":"https://github.com/takomc/amp/blob/HEAD/llava/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2404.15127","paper":"/paper/meddr-diagnosis-guided-bootstrapping-for","title":"GSCo: Towards Generalizable AI in Medicine via Generalist-Specialist Collaboration","date":"2024-04-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sunanhe/meddr","path":"src/dataset/transforms.py","file_url":"https://github.com/sunanhe/meddr/blob/HEAD/src/dataset/transforms.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2404.14233","paper":"/paper/detecting-and-mitigating-hallucination-in","title":"Detecting and Mitigating Hallucination in Large Vision Language Models via Fine-Grained AI Feedback","date":"2024-04-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Mr-Loevan/HSA-DPO","path":"hsa_dpo/models/llava-v1_5/llava/mm_utils.py","file_url":"https://github.com/Mr-Loevan/HSA-DPO/blob/HEAD/hsa_dpo/models/llava-v1_5/llava/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2404.10501","paper":"/paper/self-supervised-visual-preference-alignment","title":"Self-Supervised Visual Preference Alignment","date":"2024-04-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Kevinz-code/SeVa","path":"seva/llava/mm_utils.py","file_url":"https://github.com/Kevinz-code/SeVa/blob/HEAD/seva/llava/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"GPL-3.0","inline_ok":false,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2403.20213","paper":"/paper/h2rsvlm-towards-helpful-and-honest-remote","title":"VHM: Versatile and Honest Vision Language Model for Remote Sensing Image Analysis","date":"2024-03-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"opendatalab/h2rsvlm","path":"vhm/mm_utils.py","file_url":"https://github.com/opendatalab/h2rsvlm/blob/HEAD/vhm/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2403.18814","paper":"/paper/mini-gemini-mining-the-potential-of-multi","title":"Mini-Gemini: Mining the Potential of Multi-modality Vision Language Models","date":"2024-03-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dvlab-research/minigemini","path":"mgm/mm_utils.py","file_url":"https://github.com/dvlab-research/minigemini/blob/HEAD/mgm/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2403.08002","paper":"/paper/training-small-multimodal-models-to-bridge","title":"Towards a clinically accessible radiology foundation model: open-access and lightweight, with automated evaluation","date":"2024-03-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"microsoft/llava-rad","path":"llava/mm_utils.py","file_url":"https://github.com/microsoft/llava-rad/blob/HEAD/llava/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2403.05525","paper":"/paper/deepseek-vl-towards-real-world-vision","title":"DeepSeek-VL: Towards Real-World Vision-Language Understanding","date":"2024-03-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"deepseek-ai/deepseek-vl","path":"deepseek_vl/models/image_processing_vlm.py","file_url":"https://github.com/deepseek-ai/deepseek-vl/blob/HEAD/deepseek_vl/models/image_processing_vlm.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2403.02910","paper":"/paper/imgtrojan-jailbreaking-vision-language-models","title":"ImgTrojan: Jailbreaking Vision-Language Models with ONE Image","date":"2024-03-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xijia-tao/imgtrojan","path":"finetune/llava/mm_utils.py","file_url":"https://github.com/xijia-tao/imgtrojan/blob/HEAD/finetune/llava/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2402.03766","paper":"/paper/2402-03766","title":"MobileVLM V2: Faster and Stronger Baseline for Vision Language Model","date":"2024-02-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"meituan-automl/mobilevlm","path":"mobilevlm/utils.py","file_url":"https://github.com/meituan-automl/mobilevlm/blob/HEAD/mobilevlm/utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2401.13478","paper":"/paper/scimmir-benchmarking-scientific-multi-modal","title":"SciMMIR: Benchmarking Scientific Multi-modal Information Retrieval","date":"2024-01-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wusiwei0410/scimmir","path":"mm_utils.py","file_url":"https://github.com/wusiwei0410/scimmir/blob/HEAD/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2401.06591","paper":"/paper/prometheus-vision-vision-language-model-as-a","title":"Prometheus-Vision: Vision-Language Model as a Judge for Fine-Grained Evaluation","date":"2024-01-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kaistai/prometheus-vision","path":"llava/mm_utils.py","file_url":"https://github.com/kaistai/prometheus-vision/blob/HEAD/llava/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2401.06209","paper":"/paper/eyes-wide-shut-exploring-the-visual","title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","date":"2024-01-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tsb0601/MMVP","path":"LLaVA/llava/mm_utils.py","file_url":"https://github.com/tsb0601/MMVP/blob/HEAD/LLaVA/llava/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2401.02906","paper":"/paper/mllm-protector-ensuring-mllm-s-safety-without","title":"MLLM-Protector: Ensuring MLLM's Safety without Hurting Performance","date":"2024-01-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pipilurj/mllm-protector","path":"llava/mm_utils.py","file_url":"https://github.com/pipilurj/mllm-protector/blob/HEAD/llava/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2401.02330","paper":"/paper/llava-ph-efficient-multi-modal-assistant-with","title":"LLaVA-Phi: Efficient Multi-Modal Assistant with Small Language Model","date":"2024-01-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zhuyiche/llava-phi","path":"llava_phi/mm_utils.py","file_url":"https://github.com/zhuyiche/llava-phi/blob/HEAD/llava_phi/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2312.14233","paper":"/paper/vcoder-versatile-vision-encoders-for","title":"VCoder: Versatile Vision Encoders for Multimodal Large Language Models","date":"2023-12-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shi-labs/vcoder","path":"vcoder_llava/mm_utils.py","file_url":"https://github.com/shi-labs/vcoder/blob/HEAD/vcoder_llava/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2312.14135","paper":"/paper/textit-v-guided-visual-search-as-a-core","title":"V*: Guided Visual Search as a Core Mechanism in Multimodal LLMs","date":"2023-12-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"penghao-wu/vstar","path":"vstar_bench_eval.py","file_url":"https://github.com/penghao-wu/vstar/blob/HEAD/vstar_bench_eval.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e02a1c976f061bb2","mcp_get_code":{"code_sha256":"e02a1c976f061bb2"}},{"arxiv_id":"2312.10032","paper":"/paper/osprey-pixel-understanding-with-visual","title":"Osprey: Pixel Understanding with Visual Instruction Tuning","date":"2023-12-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"circleradon/osprey","path":"osprey/mm_utils.py","file_url":"https://github.com/circleradon/osprey/blob/HEAD/osprey/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2312.06731","paper":"/paper/genixer-empowering-multimodal-large-language","title":"Genixer: Empowering Multimodal Large Language Models as a Powerful Data Generator","date":"2023-12-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zhaohengyuan1/genixer","path":"Genixer_LLaVA/llava/mm_utils.py","file_url":"https://github.com/zhaohengyuan1/genixer/blob/HEAD/Genixer_LLaVA/llava/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2312.04746","paper":"/paper/quilt-llava-visual-instruction-tuning-by","title":"Quilt-LLaVA: Visual Instruction Tuning by Extracting Localized Narratives from Open-Source Histopathology Videos","date":"2023-12-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"aldraus/quilt-llava","path":"llava/mm_utils.py","file_url":"https://github.com/aldraus/quilt-llava/blob/HEAD/llava/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2311.16254","paper":"/paper/removing-nsfw-concepts-from-vision-and","title":"Safe-CLIP: Removing NSFW Concepts from Vision-and-Language Models","date":"2023-11-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"aimagelab/safe-clip","path":"LLaVA_generation/llava/mm_utils.py","file_url":"https://github.com/aimagelab/safe-clip/blob/HEAD/LLaVA_generation/llava/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2311.15826","paper":"/paper/geochat-grounded-large-vision-language-model","title":"GeoChat: Grounded Large Vision-Language Model for Remote Sensing","date":"2023-11-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mbzuai-oryx/geochat","path":"geochat/mm_utils.py","file_url":"https://github.com/mbzuai-oryx/geochat/blob/HEAD/geochat/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2311.13194","paper":"/paper/towards-improving-document-understanding-an","title":"Towards Improving Document Understanding: An Exploration on Text-Grounding via MLLMs","date":"2023-11-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"harrytea/tgdoc","path":"tgdoc/mm_utils.py","file_url":"https://github.com/harrytea/tgdoc/blob/HEAD/tgdoc/mm_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"8c5ec74733471314","mcp_get_code":{"code_sha256":"8c5ec74733471314"}},{"arxiv_id":"2310.17956","paper":"/paper/qilin-med-vl-towards-chinese-large-vision","title":"Qilin-Med-VL: Towards Chinese Large Vision-Language Model for General Healthcare","date":"2023-10-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"williamliujl/qilin-med-vl","path":"llava/mm_utils.py","file_url":"https://github.com/williamliujl/qilin-med-vl/blob/HEAD/llava/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2310.15110","paper":"/paper/zero123-a-single-image-to-consistent-multi","title":"Zero123++: a Single Image to Consistent Multi-view Diffusion Base Model","date":"2023-10-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sudo-ai-3d/zero123plus","path":"gradio_app.py","file_url":"https://github.com/sudo-ai-3d/zero123plus/blob/HEAD/gradio_app.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2310.09291","paper":"/paper/vision-by-language-for-training-free","title":"Vision-by-Language for Training-Free Compositional Image Retrieval","date":"2023-10-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"explainableml/vision_by_language","path":"src/datasets.py","file_url":"https://github.com/explainableml/vision_by_language/blob/HEAD/src/datasets.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e75146036d2d1504","mcp_get_code":{"code_sha256":"e75146036d2d1504"}},{"arxiv_id":"2310.01779","paper":"/paper/halle-switch-rethinking-and-controlling","title":"HallE-Control: Controlling Object Hallucination in Large Multimodal Models","date":"2023-10-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bronyayang/HallE_Switch","path":"llava/mm_utils.py","file_url":"https://github.com/bronyayang/HallE_Switch/blob/HEAD/llava/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2310.00582","paper":"/paper/pink-unveiling-the-power-of-referential","title":"Pink: Unveiling the Power of Referential Comprehension for Multi-modal LLMs","date":"2023-10-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sy-xuan/pink","path":"pink/eval/model_gqa.py","file_url":"https://github.com/sy-xuan/pink/blob/HEAD/pink/eval/model_gqa.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ee1020522a070904","mcp_get_code":{"code_sha256":"ee1020522a070904"}},{"arxiv_id":"2308.10253","paper":"/paper/stablellava-enhanced-visual-instruction","title":"StableLLaVA: Enhanced Visual Instruction Tuning with Synthesized Image-Dialogue Data","date":"2023-08-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"icoz69/stablellava","path":"llava/mm_utils.py","file_url":"https://github.com/icoz69/stablellava/blob/HEAD/llava/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2308.03060","paper":"/paper/topiq-a-top-down-approach-from-semantics-to","title":"TOPIQ: A Top-down Approach from Semantics to Distortions for Image Quality Assessment","date":"2023-08-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"chaofengc/iqa-pytorch","path":"pyiqa/archs/compare2score_arch.py","file_url":"https://github.com/chaofengc/iqa-pytorch/blob/HEAD/pyiqa/archs/compare2score_arch.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"66c27cd25f8c24da","mcp_get_code":{"code_sha256":"66c27cd25f8c24da"}},{"arxiv_id":"2204.03541","paper":"/paper/end-to-end-zero-shot-hoi-detection-via-vision","title":"End-to-End Zero-Shot HOI Detection via Vision and Language Knowledge Distillation","date":"2022-04-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mrwu-mac/EoID","path":"models/EoID.py","file_url":"https://github.com/mrwu-mac/EoID/blob/HEAD/models/EoID.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2112.05682","paper":"/paper/self-attention-does-not-need-o-n-2-memory","title":"Self-attention Does Not Need $O(n^2)$ Memory","date":"2021-12-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jihaonew/mm-instruct","path":"llava/mm_utils.py","file_url":"https://github.com/jihaonew/mm-instruct/blob/HEAD/llava/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2112.05682","paper":"/paper/self-attention-does-not-need-o-n-2-memory","title":"Self-attention Does Not Need $O(n^2)$ Memory","date":"2021-12-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"X-iZhang/Libra","path":"libra/mm_utils.py","file_url":"https://github.com/X-iZhang/Libra/blob/HEAD/libra/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"1fe910f8046380d4","mcp_get_code":{"code_sha256":"1fe910f8046380d4"}},{"arxiv_id":"Zhang_Learning_Rain_Location_Prior_for_Nighttime_Deraining_ICCV_2023_paper","paper":null,"title":"arXiv:Zhang_Learning_Rain_Location_Prior_for_Nighttime_Deraining_ICCV_2023_paper","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"zkawfanx/RLP","path":"rlp/utils.py","file_url":"https://github.com/zkawfanx/RLP/blob/HEAD/rlp/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"184826b4b0deed67","mcp_get_code":{"code_sha256":"184826b4b0deed67"}},{"arxiv_id":"2025.naacl-long.579","paper":null,"title":"arXiv:2025.naacl-long.579","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"DAMO-NLP-SG/VideoLLaMA2","path":"videollama2/mm_utils.py","file_url":"https://github.com/DAMO-NLP-SG/VideoLLaMA2/blob/HEAD/videollama2/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2024.findings-naacl.226","paper":null,"title":"arXiv:2024.findings-naacl.226","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"nguyennm1024/OSCaR","path":"llava/mm_utils.py","file_url":"https://github.com/nguyennm1024/OSCaR/blob/HEAD/llava/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2024.findings-emnlp.775","paper":null,"title":"arXiv:2024.findings-emnlp.775","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"YuxiXie/V-DPO","path":"llava_dpo/mm_utils.py","file_url":"https://github.com/YuxiXie/V-DPO/blob/HEAD/llava_dpo/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}},{"arxiv_id":"2024.findings-emnlp.268","paper":null,"title":"arXiv:2024.findings-emnlp.268","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"HZQ950419/Math-LLaVA","path":"llava/mm_utils.py","file_url":"https://github.com/HZQ950419/Math-LLaVA/blob/HEAD/llava/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"592b3c1a88f93d7c","mcp_get_code":{"code_sha256":"592b3c1a88f93d7c"}}]}