{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/process-images","entry":"process_images","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":79,"n_papers_ran":35,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":18,"n_samples_ran":10,"n_samples_fingerprinted":0,"n_places":81,"n_places_pointer_only":33,"by_status":{"ran_honours":2,"ran_violates":0,"ran_draft_wrong":1,"ran_fixture":0,"ran":7,"unverified":8},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2605.27378","paper":"/paper/arxiv-2605-27378","title":"OralAgent: Integrating Reasoning, Tools, and Knowledge for Interactive Dental Image Analysis","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"isjinghao/OralAgent","path":"oralagent/llava/mm_utils.py","file_url":"https://github.com/isjinghao/OralAgent/blob/HEAD/oralagent/llava/mm_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f0f8c9c5a77c9b5b","mcp_get_code":{"code_sha256":"f0f8c9c5a77c9b5b"}},{"arxiv_id":"2605.13178","paper":"/paper/arxiv-2605-13178","title":"CLIP Tricks You: Training-free Token Pruning for Efficient Pixel Grounding in Large Vision-Language Models","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"sejong-rcv/LiteLVLM","path":"model/llava/mm_utils.py","file_url":"https://github.com/sejong-rcv/LiteLVLM/blob/HEAD/model/llava/mm_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"1df990c375318896","mcp_get_code":{"code_sha256":"1df990c375318896"}},{"arxiv_id":"2605.11591","paper":"/paper/arxiv-2605-11591","title":"Logit-Attention Divergence: Mitigating Position Bias in Multi-Image Retrieval via Attention-Guided Calibration","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"brightXian/LAD","path":"src/agd.py","file_url":"https://github.com/brightXian/LAD/blob/HEAD/src/agd.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f53971fbc29934cc","mcp_get_code":{"code_sha256":"f53971fbc29934cc"}},{"arxiv_id":"2603.21426","paper":"/paper/arxiv-2603-21426","title":"Uncertainty-Aware Knowledge Distillation for Multimodal Large Language Models","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"Jingchensun/beta-kd","path":"mobilevlm/utils.py","file_url":"https://github.com/Jingchensun/beta-kd/blob/HEAD/mobilevlm/utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e8f3a2a1be36a4d7","mcp_get_code":{"code_sha256":"e8f3a2a1be36a4d7"}},{"arxiv_id":"2602.09483","paper":"/paper/arxiv-2602-09483","title":"Beyond Next-Token Alignment: Distilling Multimodal Large Language Models via Token Interactions","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"lchen1019/Align-TI","path":"alignti/mm_utils.py","file_url":"https://github.com/lchen1019/Align-TI/blob/HEAD/alignti/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"344dff4791fd1381","mcp_get_code":{"code_sha256":"344dff4791fd1381"}},{"arxiv_id":"2601.17918","paper":"/paper/arxiv-2601-17918","title":"Benchmarking Direct Preference Optimization for Medical Large Vision-Language Models","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"dmis-lab/med-vlm-dpo","path":"inference/LLaVA-Med/llava/mm_utils.py","file_url":"https://github.com/dmis-lab/med-vlm-dpo/blob/HEAD/inference/LLaVA-Med/llava/mm_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f0f8c9c5a77c9b5b","mcp_get_code":{"code_sha256":"f0f8c9c5a77c9b5b"}},{"arxiv_id":"2509.24365","paper":"/paper/arxiv-2509-24365","title":"Uni-X: Mitigating Modality Conflict with a Two-End-Separated Architecture for Unified Multimodal Models","date":null,"month_inferred_from_arxiv_id":"2025-09","title_source":"syntology","repo":"CURRENTF/Uni-X","path":"uni_arch/mm_utils.py","file_url":"https://github.com/CURRENTF/Uni-X/blob/HEAD/uni_arch/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d6153bc4456b4b0c","mcp_get_code":{"code_sha256":"d6153bc4456b4b0c"}},{"arxiv_id":"2505.20753","paper":"/paper/understand-think-and-answer-advancing-visual","title":"Understand, Think, and Answer: Advancing Visual Reasoning with Large Multimodal Models","date":"2025-05-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jefferyzhan/griffon","path":"griffon/mm_utils.py","file_url":"https://github.com/jefferyzhan/griffon/blob/HEAD/griffon/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"29489f5332508b33","mcp_get_code":{"code_sha256":"29489f5332508b33"}},{"arxiv_id":"2504.09644","paper":"/paper/segearth-r1-geospatial-pixel-reasoning-via","title":"SegEarth-R1: Geospatial Pixel Reasoning via Large Language Model","date":"2025-04-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"earth-insights/segearth-r1","path":"segearth_r1/mm_utils.py","file_url":"https://github.com/earth-insights/segearth-r1/blob/HEAD/segearth_r1/mm_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"1df990c375318896","mcp_get_code":{"code_sha256":"1df990c375318896"}},{"arxiv_id":"2503.21851","paper":"/paper/on-large-multimodal-models-as-open-world","title":"On Large Multimodal Models as Open-World Image Classifiers","date":"2025-03-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"altndrr/lmms-owc","path":"src/models/_instructblip.py","file_url":"https://github.com/altndrr/lmms-owc/blob/HEAD/src/models/_instructblip.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"24156f8626f7fa5d","mcp_get_code":{"code_sha256":"24156f8626f7fa5d"}},{"arxiv_id":"2502.06788","paper":"/paper/evev2-improved-baselines-for-encoder-free","title":"EVEv2: Improved Baselines for Encoder-Free Vision-Language Models","date":"2025-02-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"baaivision/EVE","path":"EVEv1/eve/mm_utils.py","file_url":"https://github.com/baaivision/EVE/blob/HEAD/EVEv1/eve/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a87d7637b728e681","mcp_get_code":{"code_sha256":"a87d7637b728e681"}},{"arxiv_id":"2502.02673","paper":"/paper/medrax-medical-reasoning-agent-for-chest-x","title":"MedRAX: Medical Reasoning Agent for Chest X-ray","date":"2025-02-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bowang-lab/medrax","path":"medrax/llava/mm_utils.py","file_url":"https://github.com/bowang-lab/medrax/blob/HEAD/medrax/llava/mm_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f0f8c9c5a77c9b5b","mcp_get_code":{"code_sha256":"f0f8c9c5a77c9b5b"}},{"arxiv_id":"2501.03218","paper":"/paper/dispider-enabling-video-llms-with-active-real","title":"Dispider: Enabling Video LLMs with Active Real-Time Interaction via Disentangled Perception, Decision, and Reaction","date":"2025-01-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Mark12Ding/Dispider","path":"dispider/mm_utils.py","file_url":"https://github.com/Mark12Ding/Dispider/blob/HEAD/dispider/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"18015a195a7c0e68","mcp_get_code":{"code_sha256":"18015a195a7c0e68"}},{"arxiv_id":"2412.14006","paper":"/paper/instructseg-unifying-instructed-visual","title":"InstructSeg: Unifying Instructed Visual Segmentation with Multi-modal Large Language Models","date":"2024-12-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"congvvc/instructseg","path":"instructseg/model/mipha/mm_utils.py","file_url":"https://github.com/congvvc/instructseg/blob/HEAD/instructseg/model/mipha/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"344dff4791fd1381","mcp_get_code":{"code_sha256":"344dff4791fd1381"}},{"arxiv_id":"2412.09951","paper":"/paper/wisead-knowledge-augmented-end-to-end","title":"WiseAD: Knowledge Augmented End-to-End Autonomous Driving with Vision-Language Model","date":"2024-12-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wyddmw/WiseAD","path":"mobilevlm/utils.py","file_url":"https://github.com/wyddmw/WiseAD/blob/HEAD/mobilevlm/utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e8f3a2a1be36a4d7","mcp_get_code":{"code_sha256":"e8f3a2a1be36a4d7"}},{"arxiv_id":"2411.19772","paper":"/paper/longvale-vision-audio-language-event","title":"LongVALE: Vision-Audio-Language-Event Benchmark Towards Time-Aware Omni-Modal Perception of Long Videos","date":"2024-11-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ttgeng233/LongVALE","path":"longvalellm/mm_utils.py","file_url":"https://github.com/ttgeng233/LongVALE/blob/HEAD/longvalellm/mm_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"1df990c375318896","mcp_get_code":{"code_sha256":"1df990c375318896"}},{"arxiv_id":"2411.17606","paper":"/paper/hyperseg-towards-universal-visual","title":"HyperSeg: Towards Universal Visual Segmentation with Large Language Model","date":"2024-11-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"congvvc/HyperSeg","path":"hyperseg/model/mipha/mm_utils.py","file_url":"https://github.com/congvvc/HyperSeg/blob/HEAD/hyperseg/model/mipha/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"344dff4791fd1381","mcp_get_code":{"code_sha256":"344dff4791fd1381"}},{"arxiv_id":"2411.02712","paper":"/paper/v-dpo-mitigating-hallucination-in-large","title":"V-DPO: Mitigating Hallucination in Large Vision Language Models via Vision-Guided Direct Preference Optimization","date":"2024-11-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yuxixie/v-dpo","path":"llava_dpo/mm_utils.py","file_url":"https://github.com/yuxixie/v-dpo/blob/HEAD/llava_dpo/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"344dff4791fd1381","mcp_get_code":{"code_sha256":"344dff4791fd1381"}},{"arxiv_id":"2410.04780","paper":"/paper/mitigating-modality-prior-induced","title":"Mitigating Modality Prior-Induced Hallucinations in Multimodal Large Language Models via Deciphering Attention Causality","date":"2024-10-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"The-Martyr/CausalMM","path":"llava-1.5/experiments/llava/mm_utils.py","file_url":"https://github.com/The-Martyr/CausalMM/blob/HEAD/llava-1.5/experiments/llava/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"344dff4791fd1381","mcp_get_code":{"code_sha256":"344dff4791fd1381"}},{"arxiv_id":"2409.17143","paper":"/paper/attention-prompting-on-image-for-large-vision","title":"Attention Prompting on Image for Large Vision-Language Models","date":"2024-09-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yu-rp/apiprompting","path":"API/API_LLaVA/functions.py","file_url":"https://github.com/yu-rp/apiprompting/blob/HEAD/API/API_LLaVA/functions.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e8f3a2a1be36a4d7","mcp_get_code":{"code_sha256":"e8f3a2a1be36a4d7"}},{"arxiv_id":"2409.16261","paper":"/paper/cdchat-a-large-multimodal-model-for-remote","title":"CDChat: A Large Multimodal Model for Remote Sensing Change Description","date":"2024-09-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"techmn/cdchat","path":"cdchat/mm_utils.py","file_url":"https://github.com/techmn/cdchat/blob/HEAD/cdchat/mm_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ab382f4b6886dee2","mcp_get_code":{"code_sha256":"ab382f4b6886dee2"}},{"arxiv_id":"2409.15657","paper":"/paper/mmpt-multimodal-prompt-tuning-for-zero-shot","title":"M$^2$PT: Multimodal Prompt Tuning for Zero-shot Instruction Learning","date":"2024-09-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"william-wang618/mmpt-emnlp2024","path":"M2PT/mm_utils.py","file_url":"https://github.com/william-wang618/mmpt-emnlp2024/blob/HEAD/M2PT/mm_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1df990c375318896","mcp_get_code":{"code_sha256":"1df990c375318896"}},{"arxiv_id":"2409.10197","paper":"/paper/fit-and-prune-fast-and-training-free-visual","title":"Fit and Prune: Fast and Training-free Visual Token Pruning for Multi-modal Large Language Models","date":"2024-09-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ywh187/fitprune","path":"LLaVA_1.5/llava/mm_utils.py","file_url":"https://github.com/ywh187/fitprune/blob/HEAD/LLaVA_1.5/llava/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"344dff4791fd1381","mcp_get_code":{"code_sha256":"344dff4791fd1381"}},{"arxiv_id":"2409.08582","paper":"/paper/changechat-an-interactive-model-for-remote","title":"ChangeChat: An Interactive Model for Remote Sensing Change Analysis via Multimodal Instruction Tuning","date":"2024-09-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hanlinwu/changechat","path":"changechat/mm_utils.py","file_url":"https://github.com/hanlinwu/changechat/blob/HEAD/changechat/mm_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bec15ff45f271129","mcp_get_code":{"code_sha256":"bec15ff45f271129"}},{"arxiv_id":"2409.01179","paper":"/paper/recoverable-compression-a-multimodal-vision","title":"Recoverable Compression: A Multimodal Vision Token Recovery Mechanism Guided by Text Information","date":"2024-09-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"banjiuyufen/Recoverable-Compression","path":"llava/mm_utils.py","file_url":"https://github.com/banjiuyufen/Recoverable-Compression/blob/HEAD/llava/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"344dff4791fd1381","mcp_get_code":{"code_sha256":"344dff4791fd1381"}},{"arxiv_id":"2408.16944","paper":"/paper/flowretrieval-flow-guided-data-retrieval-for","title":"FlowRetrieval: Flow-Guided Data Retrieval for Few-Shot Imitation Learning","date":"2024-08-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lihenglin/bridge_training_code","path":"data_processing/bridgedata_raw_to_numpy.py","file_url":"https://github.com/lihenglin/bridge_training_code/blob/HEAD/data_processing/bridgedata_raw_to_numpy.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d2537349eee21015","mcp_get_code":{"code_sha256":"d2537349eee21015"}},{"arxiv_id":"2408.15992","paper":"/paper/cogen-learning-from-feedback-with-coupled","title":"CoGen: Learning from Feedback with Coupled Comprehension and Generation","date":"2024-08-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lil-lab/cogen","path":"models/joint_inference.py","file_url":"https://github.com/lil-lab/cogen/blob/HEAD/models/joint_inference.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"5f621459c8e0f985","mcp_get_code":{"code_sha256":"5f621459c8e0f985"}},{"arxiv_id":"2408.12902","paper":"/paper/iaa-inner-adaptor-architecture-empowers","title":"IAA: Inner-Adaptor Architecture Empowers Frozen Large Language Model with Multimodal Capabilities","date":"2024-08-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"360cvgroup/inner-adaptor-architecture","path":"iaa/mm_utils.py","file_url":"https://github.com/360cvgroup/inner-adaptor-architecture/blob/HEAD/iaa/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"344dff4791fd1381","mcp_get_code":{"code_sha256":"344dff4791fd1381"}},{"arxiv_id":"2407.21771","paper":"/paper/paying-more-attention-to-image-a-training","title":"Paying More Attention to Image: A Training-Free Method for Alleviating Hallucination in LVLMs","date":"2024-07-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hasanar1f/llava-hallunication-fix","path":"modPAI/llava/mm_utils.py","file_url":"https://github.com/hasanar1f/llava-hallunication-fix/blob/HEAD/modPAI/llava/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"344dff4791fd1381","mcp_get_code":{"code_sha256":"344dff4791fd1381"}},{"arxiv_id":"2407.08303","paper":"/paper/densefusion-1m-merging-vision-experts-for","title":"DenseFusion-1M: Merging Vision Experts for Comprehensive Multimodal Perception","date":"2024-07-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"baaivision/DenseFusion","path":"densefusion/mm_utils.py","file_url":"https://github.com/baaivision/DenseFusion/blob/HEAD/densefusion/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e8f3a2a1be36a4d7","mcp_get_code":{"code_sha256":"e8f3a2a1be36a4d7"}},{"arxiv_id":"2406.20098","paper":"/paper/web2code-a-large-scale-webpage-to-code","title":"Web2Code: A Large-scale Webpage-to-Code Dataset and Evaluation Framework for Multimodal LLMs","date":"2024-06-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"MBZUAI-LLM/web2code","path":"web2code/llava/mm_utils.py","file_url":"https://github.com/MBZUAI-LLM/web2code/blob/HEAD/web2code/llava/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"344dff4791fd1381","mcp_get_code":{"code_sha256":"344dff4791fd1381"}},{"arxiv_id":"2406.13173","paper":"/paper/biomedical-visual-instruction-tuning-with","title":"Biomedical Visual Instruction Tuning with Clinician Preference Alignment","date":"2024-06-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mao1207/BioMed-VITAL","path":"backbone/mm_utils.py","file_url":"https://github.com/mao1207/BioMed-VITAL/blob/HEAD/backbone/mm_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f0f8c9c5a77c9b5b","mcp_get_code":{"code_sha256":"f0f8c9c5a77c9b5b"}},{"arxiv_id":"2406.12235","paper":"/paper/holmes-vad-towards-unbiased-and-explainable","title":"Holmes-VAD: Towards Unbiased and Explainable Video Anomaly Detection via Multi-modal LLM","date":"2024-06-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pipixin321/holmesvad","path":"videollava/mm_utils.py","file_url":"https://github.com/pipixin321/holmesvad/blob/HEAD/videollava/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"344dff4791fd1381","mcp_get_code":{"code_sha256":"344dff4791fd1381"}},{"arxiv_id":"2406.11839","paper":"/paper/mdpo-conditional-preference-optimization-for","title":"mDPO: Conditional Preference Optimization for Multimodal Large Language Models","date":"2024-06-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"luka-group/mDPO","path":"bunny/bunny_utils/util/mm_utils.py","file_url":"https://github.com/luka-group/mDPO/blob/HEAD/bunny/bunny_utils/util/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e8f3a2a1be36a4d7","mcp_get_code":{"code_sha256":"e8f3a2a1be36a4d7"}},{"arxiv_id":"2406.10215","paper":"/paper/devbench-a-multimodal-developmental-benchmark","title":"DevBench: A multimodal developmental benchmark for language learning","date":"2024-06-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"alvinwmtan/dev-bench","path":"model_classes/modeling_tinyllava_phi.py","file_url":"https://github.com/alvinwmtan/dev-bench/blob/HEAD/model_classes/modeling_tinyllava_phi.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"58a932ffe36876f1","mcp_get_code":{"code_sha256":"58a932ffe36876f1"}},{"arxiv_id":"2406.09418","paper":"/paper/videogpt-integrating-image-and-video-encoders","title":"VideoGPT+: Integrating Image and Video Encoders for Enhanced Video Understanding","date":"2024-06-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mbzuai-oryx/videogpt-plus","path":"videogpt_plus/mm_utils.py","file_url":"https://github.com/mbzuai-oryx/videogpt-plus/blob/HEAD/videogpt_plus/mm_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"CC-BY-4.0","inline_ok":false,"code_sha256_prefix":"1df990c375318896","mcp_get_code":{"code_sha256":"1df990c375318896"}},{"arxiv_id":"2406.08100","paper":"/paper/multimodal-table-understanding","title":"Multimodal Table Understanding","date":"2024-06-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"spursgozmy/table-llava","path":"llava/mm_utils.py","file_url":"https://github.com/spursgozmy/table-llava/blob/HEAD/llava/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"344dff4791fd1381","mcp_get_code":{"code_sha256":"344dff4791fd1381"}},{"arxiv_id":"2406.02884","paper":"/paper/posterllava-constructing-a-unified-multi","title":"PosterLLaVa: Constructing a Unified Multi-modal Layout Generator with LLM","date":"2024-06-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"posterllava/posterllava","path":"llava/mm_utils.py","file_url":"https://github.com/posterllava/posterllava/blob/HEAD/llava/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"344dff4791fd1381","mcp_get_code":{"code_sha256":"344dff4791fd1381"}},{"arxiv_id":"2405.19298","paper":"/paper/adaptive-image-quality-assessment-via","title":"Adaptive Image Quality Assessment via Teaching Large Multimodal Model to Compare","date":"2024-05-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Q-Future/Compare2Score","path":"q_align/mm_utils.py","file_url":"https://github.com/Q-Future/Compare2Score/blob/HEAD/q_align/mm_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"af0990e5a0f15a4d","mcp_get_code":{"code_sha256":"af0990e5a0f15a4d"}},{"arxiv_id":"2405.14974","paper":"/paper/lova3-learning-to-visual-question-answering","title":"LOVA3: Learning to Visual Question Answering, Asking and Assessment","date":"2024-05-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"showlab/LOVA3","path":"llava/mm_utils.py","file_url":"https://github.com/showlab/LOVA3/blob/HEAD/llava/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"344dff4791fd1381","mcp_get_code":{"code_sha256":"344dff4791fd1381"}},{"arxiv_id":"2405.12107","paper":"/paper/imp-highly-capable-large-multimodal-models","title":"Imp: Highly Capable Large Multimodal Models for Mobile Devices","date":"2024-05-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"milvlg/imp","path":"imp_llava/mm_utils.py","file_url":"https://github.com/milvlg/imp/blob/HEAD/imp_llava/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"344dff4791fd1381","mcp_get_code":{"code_sha256":"344dff4791fd1381"}},{"arxiv_id":"2405.11165","paper":"/paper/automated-multi-level-preference-for-mllms","title":"Automated Multi-level Preference for MLLMs","date":"2024-05-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"takomc/amp","path":"llava/mm_utils.py","file_url":"https://github.com/takomc/amp/blob/HEAD/llava/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"344dff4791fd1381","mcp_get_code":{"code_sha256":"344dff4791fd1381"}},{"arxiv_id":"2404.14233","paper":"/paper/detecting-and-mitigating-hallucination-in","title":"Detecting and Mitigating Hallucination in Large Vision Language Models via Fine-Grained AI Feedback","date":"2024-04-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Mr-Loevan/HSA-DPO","path":"hsa_dpo/models/llava-v1_5/llava/mm_utils.py","file_url":"https://github.com/Mr-Loevan/HSA-DPO/blob/HEAD/hsa_dpo/models/llava-v1_5/llava/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"344dff4791fd1381","mcp_get_code":{"code_sha256":"344dff4791fd1381"}},{"arxiv_id":"2404.10501","paper":"/paper/self-supervised-visual-preference-alignment","title":"Self-Supervised Visual Preference Alignment","date":"2024-04-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Kevinz-code/SeVa","path":"seva/llava/mm_utils.py","file_url":"https://github.com/Kevinz-code/SeVa/blob/HEAD/seva/llava/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"GPL-3.0","inline_ok":false,"code_sha256_prefix":"344dff4791fd1381","mcp_get_code":{"code_sha256":"344dff4791fd1381"}},{"arxiv_id":"2404.08506","paper":"/paper/lasagna-language-based-segmentation-assistant","title":"LaSagnA: Language-based Segmentation Assistant for Complex Queries","date":"2024-04-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"congvvc/lasagna","path":"model/llava/mm_utils.py","file_url":"https://github.com/congvvc/lasagna/blob/HEAD/model/llava/mm_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"1df990c375318896","mcp_get_code":{"code_sha256":"1df990c375318896"}},{"arxiv_id":"2403.20213","paper":"/paper/h2rsvlm-towards-helpful-and-honest-remote","title":"VHM: Versatile and Honest Vision Language Model for Remote Sensing Image Analysis","date":"2024-03-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"opendatalab/h2rsvlm","path":"vhm/mm_utils.py","file_url":"https://github.com/opendatalab/h2rsvlm/blob/HEAD/vhm/mm_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"e8f3a2a1be36a4d7","mcp_get_code":{"code_sha256":"e8f3a2a1be36a4d7"}},{"arxiv_id":"2403.18814","paper":"/paper/mini-gemini-mining-the-potential-of-multi","title":"Mini-Gemini: Mining the Potential of Multi-modality Vision Language Models","date":"2024-03-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dvlab-research/minigemini","path":"mgm/mm_utils.py","file_url":"https://github.com/dvlab-research/minigemini/blob/HEAD/mgm/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"d6153bc4456b4b0c","mcp_get_code":{"code_sha256":"d6153bc4456b4b0c"}},{"arxiv_id":"2403.14598","paper":"/paper/psalm-pixelwise-segmentation-with-large-multi","title":"PSALM: Pixelwise SegmentAtion with Large Multi-Modal Model","date":"2024-03-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zamling/PSALM","path":"psalm/mm_utils.py","file_url":"https://github.com/zamling/PSALM/blob/HEAD/psalm/mm_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"1df990c375318896","mcp_get_code":{"code_sha256":"1df990c375318896"}},{"arxiv_id":"2403.08002","paper":"/paper/training-small-multimodal-models-to-bridge","title":"Towards a clinically accessible radiology foundation model: open-access and lightweight, with automated evaluation","date":"2024-03-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"microsoft/llava-rad","path":"llava/mm_utils.py","file_url":"https://github.com/microsoft/llava-rad/blob/HEAD/llava/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"344dff4791fd1381","mcp_get_code":{"code_sha256":"344dff4791fd1381"}},{"arxiv_id":"2403.04640","paper":"/paper/cat-enhancing-multimodal-large-language-model","title":"CAT: Enhancing Multimodal Large Language Model to Answer Questions in Dynamic Audio-Visual Scenarios","date":"2024-03-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rikeilong/bay-cat","path":"ADPO_CAT/mm_utils.py","file_url":"https://github.com/rikeilong/bay-cat/blob/HEAD/ADPO_CAT/mm_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"1df990c375318896","mcp_get_code":{"code_sha256":"1df990c375318896"}},{"arxiv_id":"2403.02910","paper":"/paper/imgtrojan-jailbreaking-vision-language-models","title":"ImgTrojan: Jailbreaking Vision-Language Models with ONE Image","date":"2024-03-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xijia-tao/imgtrojan","path":"finetune/llava/mm_utils.py","file_url":"https://github.com/xijia-tao/imgtrojan/blob/HEAD/finetune/llava/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"344dff4791fd1381","mcp_get_code":{"code_sha256":"344dff4791fd1381"}},{"arxiv_id":"2402.11941","paper":"/paper/comprehensive-cognitive-llm-agent-for","title":"CoCo-Agent: A Comprehensive Cognitive MLLM Agent for Smartphone GUI Automation","date":"2024-02-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xbmxb/coco-agent","path":"llava/mm_utils.py","file_url":"https://github.com/xbmxb/coco-agent/blob/HEAD/llava/mm_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1df990c375318896","mcp_get_code":{"code_sha256":"1df990c375318896"}},{"arxiv_id":"2402.03766","paper":"/paper/2402-03766","title":"MobileVLM V2: Faster and Stronger Baseline for Vision Language Model","date":"2024-02-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"meituan-automl/mobilevlm","path":"mobilevlm/utils.py","file_url":"https://github.com/meituan-automl/mobilevlm/blob/HEAD/mobilevlm/utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"e8f3a2a1be36a4d7","mcp_get_code":{"code_sha256":"e8f3a2a1be36a4d7"}},{"arxiv_id":"2401.13478","paper":"/paper/scimmir-benchmarking-scientific-multi-modal","title":"SciMMIR: Benchmarking Scientific Multi-modal Information Retrieval","date":"2024-01-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wusiwei0410/scimmir","path":"mm_utils.py","file_url":"https://github.com/wusiwei0410/scimmir/blob/HEAD/mm_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"af0990e5a0f15a4d","mcp_get_code":{"code_sha256":"af0990e5a0f15a4d"}},{"arxiv_id":"2401.13478","paper":"/paper/scimmir-benchmarking-scientific-multi-modal","title":"SciMMIR: Benchmarking Scientific Multi-modal Information Retrieval","date":"2024-01-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wusiwei0410/scimmir","path":"LLMs_Embedding.py","file_url":"https://github.com/wusiwei0410/scimmir/blob/HEAD/LLMs_Embedding.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"7278092ebcd87065","mcp_get_code":{"code_sha256":"7278092ebcd87065"}},{"arxiv_id":"2401.06591","paper":"/paper/prometheus-vision-vision-language-model-as-a","title":"Prometheus-Vision: Vision-Language Model as a Judge for Fine-Grained Evaluation","date":"2024-01-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kaistai/prometheus-vision","path":"llava/mm_utils.py","file_url":"https://github.com/kaistai/prometheus-vision/blob/HEAD/llava/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"344dff4791fd1381","mcp_get_code":{"code_sha256":"344dff4791fd1381"}},{"arxiv_id":"2401.06209","paper":"/paper/eyes-wide-shut-exploring-the-visual","title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","date":"2024-01-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tsb0601/MMVP","path":"LLaVA/llava/mm_utils.py","file_url":"https://github.com/tsb0601/MMVP/blob/HEAD/LLaVA/llava/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"344dff4791fd1381","mcp_get_code":{"code_sha256":"344dff4791fd1381"}},{"arxiv_id":"2401.02906","paper":"/paper/mllm-protector-ensuring-mllm-s-safety-without","title":"MLLM-Protector: Ensuring MLLM's Safety without Hurting Performance","date":"2024-01-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pipilurj/mllm-protector","path":"llava/mm_utils.py","file_url":"https://github.com/pipilurj/mllm-protector/blob/HEAD/llava/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"344dff4791fd1381","mcp_get_code":{"code_sha256":"344dff4791fd1381"}},{"arxiv_id":"2401.02330","paper":"/paper/llava-ph-efficient-multi-modal-assistant-with","title":"LLaVA-Phi: Efficient Multi-Modal Assistant with Small Language Model","date":"2024-01-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zhuyiche/llava-phi","path":"llava_phi/mm_utils.py","file_url":"https://github.com/zhuyiche/llava-phi/blob/HEAD/llava_phi/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"344dff4791fd1381","mcp_get_code":{"code_sha256":"344dff4791fd1381"}},{"arxiv_id":"2312.17240","paper":"/paper/an-improved-baseline-for-reasoning","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","date":"2023-12-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dvlab-research/lisa","path":"model/llava/mm_utils.py","file_url":"https://github.com/dvlab-research/lisa/blob/HEAD/model/llava/mm_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"1df990c375318896","mcp_get_code":{"code_sha256":"1df990c375318896"}},{"arxiv_id":"2312.14233","paper":"/paper/vcoder-versatile-vision-encoders-for","title":"VCoder: Versatile Vision Encoders for Multimodal Large Language Models","date":"2023-12-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shi-labs/vcoder","path":"vcoder_llava/mm_utils.py","file_url":"https://github.com/shi-labs/vcoder/blob/HEAD/vcoder_llava/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"344dff4791fd1381","mcp_get_code":{"code_sha256":"344dff4791fd1381"}},{"arxiv_id":"2312.10103","paper":"/paper/gsva-generalized-segmentation-via-multimodal","title":"GSVA: Generalized Segmentation via Multimodal Large Language Models","date":"2023-12-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"leaplabthu/gsva","path":"model/llava/mm_utils.py","file_url":"https://github.com/leaplabthu/gsva/blob/HEAD/model/llava/mm_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"1df990c375318896","mcp_get_code":{"code_sha256":"1df990c375318896"}},{"arxiv_id":"2312.10032","paper":"/paper/osprey-pixel-understanding-with-visual","title":"Osprey: Pixel Understanding with Visual Instruction Tuning","date":"2023-12-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"circleradon/osprey","path":"osprey/mm_utils.py","file_url":"https://github.com/circleradon/osprey/blob/HEAD/osprey/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"344dff4791fd1381","mcp_get_code":{"code_sha256":"344dff4791fd1381"}},{"arxiv_id":"2312.06731","paper":"/paper/genixer-empowering-multimodal-large-language","title":"Genixer: Empowering Multimodal Large Language Models as a Powerful Data Generator","date":"2023-12-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zhaohengyuan1/genixer","path":"Genixer_LLaVA/llava/mm_utils.py","file_url":"https://github.com/zhaohengyuan1/genixer/blob/HEAD/Genixer_LLaVA/llava/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"344dff4791fd1381","mcp_get_code":{"code_sha256":"344dff4791fd1381"}},{"arxiv_id":"2312.04746","paper":"/paper/quilt-llava-visual-instruction-tuning-by","title":"Quilt-LLaVA: Visual Instruction Tuning by Extracting Localized Narratives from Open-Source Histopathology Videos","date":"2023-12-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"aldraus/quilt-llava","path":"llava/mm_utils.py","file_url":"https://github.com/aldraus/quilt-llava/blob/HEAD/llava/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"344dff4791fd1381","mcp_get_code":{"code_sha256":"344dff4791fd1381"}},{"arxiv_id":"2312.02949","paper":"/paper/llava-grounding-grounded-visual-chat-with","title":"LLaVA-Grounding: Grounded Visual Chat with Large Multimodal Models","date":"2023-12-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ux-decoder/llava-grounding","path":"llava/mm_utils.py","file_url":"https://github.com/ux-decoder/llava-grounding/blob/HEAD/llava/mm_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"1df990c375318896","mcp_get_code":{"code_sha256":"1df990c375318896"}},{"arxiv_id":"2311.18445","paper":"/paper/vtimellm-empower-llm-to-grasp-video-moments","title":"VTimeLLM: Empower LLM to Grasp Video Moments","date":"2023-11-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"huangb23/vtimellm","path":"vtimellm/mm_utils.py","file_url":"https://github.com/huangb23/vtimellm/blob/HEAD/vtimellm/mm_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1df990c375318896","mcp_get_code":{"code_sha256":"1df990c375318896"}},{"arxiv_id":"2311.16254","paper":"/paper/removing-nsfw-concepts-from-vision-and","title":"Safe-CLIP: Removing NSFW Concepts from Vision-and-Language Models","date":"2023-11-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"aimagelab/safe-clip","path":"LLaVA_generation/llava/mm_utils.py","file_url":"https://github.com/aimagelab/safe-clip/blob/HEAD/LLaVA_generation/llava/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"344dff4791fd1381","mcp_get_code":{"code_sha256":"344dff4791fd1381"}},{"arxiv_id":"2311.16208","paper":"/paper/instructmol-multi-modal-integration-for","title":"InstructMol: Multi-Modal Integration for Building a Versatile and Reliable Molecular Assistant in Drug Discovery","date":"2023-11-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"idea-xl/instructmol","path":"llava/mm_utils.py","file_url":"https://github.com/idea-xl/instructmol/blob/HEAD/llava/mm_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"1df990c375318896","mcp_get_code":{"code_sha256":"1df990c375318896"}},{"arxiv_id":"2311.15826","paper":"/paper/geochat-grounded-large-vision-language-model","title":"GeoChat: Grounded Large Vision-Language Model for Remote Sensing","date":"2023-11-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mbzuai-oryx/geochat","path":"geochat/mm_utils.py","file_url":"https://github.com/mbzuai-oryx/geochat/blob/HEAD/geochat/mm_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bec15ff45f271129","mcp_get_code":{"code_sha256":"bec15ff45f271129"}},{"arxiv_id":"2310.17956","paper":"/paper/qilin-med-vl-towards-chinese-large-vision","title":"Qilin-Med-VL: Towards Chinese Large Vision-Language Model for General Healthcare","date":"2023-10-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"williamliujl/qilin-med-vl","path":"llava/mm_utils.py","file_url":"https://github.com/williamliujl/qilin-med-vl/blob/HEAD/llava/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"344dff4791fd1381","mcp_get_code":{"code_sha256":"344dff4791fd1381"}},{"arxiv_id":"2310.01779","paper":"/paper/halle-switch-rethinking-and-controlling","title":"HallE-Control: Controlling Object Hallucination in Large Multimodal Models","date":"2023-10-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bronyayang/HallE_Switch","path":"llava/mm_utils.py","file_url":"https://github.com/bronyayang/HallE_Switch/blob/HEAD/llava/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"344dff4791fd1381","mcp_get_code":{"code_sha256":"344dff4791fd1381"}},{"arxiv_id":"2309.07120","paper":"/paper/sight-beyond-text-multi-modal-training","title":"Sight Beyond Text: Multi-Modal Training Enhances LLMs in Truthfulness and Ethics","date":"2023-09-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ucsc-vlaa/sight-beyond-text","path":"llava/mm_utils.py","file_url":"https://github.com/ucsc-vlaa/sight-beyond-text/blob/HEAD/llava/mm_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"1df990c375318896","mcp_get_code":{"code_sha256":"1df990c375318896"}},{"arxiv_id":"2308.12952","paper":"/paper/bridgedata-v2-a-dataset-for-robot-learning-at","title":"BridgeData V2: A Dataset for Robot Learning at Scale","date":"2023-08-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rail-berkeley/BridgeData-V2","path":"data_processing/bridgedata_raw_to_numpy.py","file_url":"https://github.com/rail-berkeley/BridgeData-V2/blob/HEAD/data_processing/bridgedata_raw_to_numpy.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d2537349eee21015","mcp_get_code":{"code_sha256":"d2537349eee21015"}},{"arxiv_id":"2308.10253","paper":"/paper/stablellava-enhanced-visual-instruction","title":"StableLLaVA: Enhanced Visual Instruction Tuning with Synthesized Image-Dialogue Data","date":"2023-08-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"icoz69/stablellava","path":"llava/mm_utils.py","file_url":"https://github.com/icoz69/stablellava/blob/HEAD/llava/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"344dff4791fd1381","mcp_get_code":{"code_sha256":"344dff4791fd1381"}},{"arxiv_id":"2112.05682","paper":"/paper/self-attention-does-not-need-o-n-2-memory","title":"Self-attention Does Not Need $O(n^2)$ Memory","date":"2021-12-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jihaonew/mm-instruct","path":"llava/mm_utils.py","file_url":"https://github.com/jihaonew/mm-instruct/blob/HEAD/llava/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"344dff4791fd1381","mcp_get_code":{"code_sha256":"344dff4791fd1381"}},{"arxiv_id":"2112.05682","paper":"/paper/self-attention-does-not-need-o-n-2-memory","title":"Self-attention Does Not Need $O(n^2)$ Memory","date":"2021-12-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"X-iZhang/Libra","path":"libra/mm_utils.py","file_url":"https://github.com/X-iZhang/Libra/blob/HEAD/libra/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"012fae59ee4131ba","mcp_get_code":{"code_sha256":"012fae59ee4131ba"}},{"arxiv_id":"Xia_GSVA_Generalized_Segmentation_via_Multimodal_Large_Language_Models_CVPR_2024_paper","paper":null,"title":"arXiv:Xia_GSVA_Generalized_Segmentation_via_Multimodal_Large_Language_Models_CVPR_2024_paper","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"LeapLabTHU/GSVA","path":"model/llava/mm_utils.py","file_url":"https://github.com/LeapLabTHU/GSVA/blob/HEAD/model/llava/mm_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"1df990c375318896","mcp_get_code":{"code_sha256":"1df990c375318896"}},{"arxiv_id":"2024.findings-naacl.226","paper":null,"title":"arXiv:2024.findings-naacl.226","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"nguyennm1024/OSCaR","path":"llava/mm_utils.py","file_url":"https://github.com/nguyennm1024/OSCaR/blob/HEAD/llava/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"344dff4791fd1381","mcp_get_code":{"code_sha256":"344dff4791fd1381"}},{"arxiv_id":"2024.findings-emnlp.775","paper":null,"title":"arXiv:2024.findings-emnlp.775","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"YuxiXie/V-DPO","path":"llava_dpo/mm_utils.py","file_url":"https://github.com/YuxiXie/V-DPO/blob/HEAD/llava_dpo/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"344dff4791fd1381","mcp_get_code":{"code_sha256":"344dff4791fd1381"}},{"arxiv_id":"2024.findings-emnlp.268","paper":null,"title":"arXiv:2024.findings-emnlp.268","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"HZQ950419/Math-LLaVA","path":"llava/mm_utils.py","file_url":"https://github.com/HZQ950419/Math-LLaVA/blob/HEAD/llava/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"344dff4791fd1381","mcp_get_code":{"code_sha256":"344dff4791fd1381"}}]}