{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/is-none","entry":"is_none","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":88,"n_papers_ran":88,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":4,"n_samples_ran":4,"n_samples_fingerprinted":0,"n_places":88,"n_places_pointer_only":24,"by_status":{"ran_honours":0,"ran_violates":1,"ran_draft_wrong":0,"ran_fixture":0,"ran":3,"unverified":0},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2609.00231","paper":"/paper/arxiv-2609-00231","title":"Beyond Language Priors: Diagnosing and Fixing Visual-Origin Hallucinations in Multimodal LLM","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"zxp555/ACFT_MM26","path":"ACFT/llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/zxp555/ACFT_MM26/blob/HEAD/ACFT/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2608.04496","paper":"/paper/arxiv-2608-04496","title":"DIVE: Dynamic Iterative Visual Evidence Construction for Efficient Vision-Language Models","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"Zhong-Chenchen/DIVE","path":"llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/Zhong-Chenchen/DIVE/blob/HEAD/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2607.21936","paper":"/paper/arxiv-2607-21936","title":"Leveraging External Knowledge for Historical Document Restoration via Retrieval-Augmented Large Language Models","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"rapidfuzz/RapidFuzz","path":"src/rapidfuzz/_utils.py","file_url":"https://github.com/rapidfuzz/RapidFuzz/blob/HEAD/src/rapidfuzz/_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"eabd22370da53c9e","mcp_get_code":{"code_sha256":"eabd22370da53c9e"}},{"arxiv_id":"2607.15299","paper":"/paper/arxiv-2607-15299","title":"MLLM-DataEngine: Closing the Loop of Multimodal Instruction Tuning Data Generation","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"opendatalab/MLLM-DataEngine","path":"LLaVA/llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/opendatalab/MLLM-DataEngine/blob/HEAD/LLaVA/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2603.21426","paper":"/paper/arxiv-2603-21426","title":"Uncertainty-Aware Knowledge Distillation for Multimodal Large Language Models","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"Jingchensun/beta-kd","path":"mobilevlm/eval/model_vqa_mmbench.py","file_url":"https://github.com/Jingchensun/beta-kd/blob/HEAD/mobilevlm/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2602.23699","paper":"/paper/arxiv-2602-23699","title":"HiDrop: Hierarchical Vision Token Reduction in MLLMs via Late Injection, Concave Pyramid Pruning, and Early Exit","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"EIT-NLP/HiDrop","path":"llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/EIT-NLP/HiDrop/blob/HEAD/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2602.09483","paper":"/paper/arxiv-2602-09483","title":"Beyond Next-Token Alignment: Distilling Multimodal Large Language Models via Token Interactions","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"lchen1019/Align-TI","path":"alignti/eval/model_vqa_mmbench.py","file_url":"https://github.com/lchen1019/Align-TI/blob/HEAD/alignti/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2507.00505","paper":"/paper/llava-sp-enhancing-visual-representation-with","title":"LLaVA-SP: Enhancing Visual Representation with Visual Spatial Tokens for MLLMs","date":"2025-07-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"CnFaker/LLaVA-SP","path":"llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/CnFaker/LLaVA-SP/blob/HEAD/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2506.10967","paper":"/paper/beyond-attention-or-similarity-maximizing","title":"Beyond Attention or Similarity: Maximizing Conditional Diversity for Token Pruning in MLLMs","date":"2025-06-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"theia-4869/cdpruner","path":"llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/theia-4869/cdpruner/blob/HEAD/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2505.01456","paper":"/paper/unlearning-sensitive-information-in","title":"Unlearning Sensitive Information in Multimodal LLMs: Benchmark and Attack-Defense Evaluation","date":"2025-05-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"vaidehi99/unlok-vqa","path":"LLaVA/llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/vaidehi99/unlok-vqa/blob/HEAD/LLaVA/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2504.00502","paper":null,"title":"arXiv:2504.00502","date":null,"month_inferred_from_arxiv_id":"2025-04","title_source":null,"repo":"icip-cas/ShortV","path":"llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/icip-cas/ShortV/blob/HEAD/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2503.11832","paper":"/paper/safety-mirage-how-spurious-correlations","title":"Safety Mirage: How Spurious Correlations Undermine VLM Safety Fine-tuning","date":"2025-03-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"OPTML-Group/VLM-Safety-Unlearn","path":"llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/OPTML-Group/VLM-Safety-Unlearn/blob/HEAD/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2502.11494","paper":"/paper/stop-looking-for-important-tokens-in","title":"Stop Looking for Important Tokens in Multimodal Language Models: Duplication Matters More","date":"2025-02-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zichenwen1/dart","path":"llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/zichenwen1/dart/blob/HEAD/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2502.06884","paper":"/paper/learning-conformal-abstention-policies-for","title":"Learning Conformal Abstention Policies for Adaptive Risk Management in Large Language and Vision-Language Models","date":"2025-02-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sinatayebati/vlm-uncertainty","path":"data_utils/common_utils.py","file_url":"https://github.com/sinatayebati/vlm-uncertainty/blob/HEAD/data_utils/common_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"25593cfba3f7b483","mcp_get_code":{"code_sha256":"25593cfba3f7b483"}},{"arxiv_id":"2502.06788","paper":"/paper/evev2-improved-baselines-for-encoder-free","title":"EVEv2: Improved Baselines for Encoder-Free Vision-Language Models","date":"2025-02-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"baaivision/EVE","path":"EVEv1/eve/eval/model_vqa_mmbench.py","file_url":"https://github.com/baaivision/EVE/blob/HEAD/EVEv1/eve/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2412.05819","paper":"/paper/cls-token-tells-everything-needed-for","title":"[CLS] Token Tells Everything Needed for Training-free Efficient MLLMs","date":"2024-12-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thu-mig/vtc-cls","path":"llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/thu-mig/vtc-cls/blob/HEAD/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2412.05255","paper":"/paper/teamcraft-a-benchmark-for-multi-modal-multi","title":"TeamCraft: A Benchmark for Multi-Modal Multi-Agent Systems in Minecraft","date":"2024-12-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"teamcraft-bench/teamcraft","path":"llava_teamcraft/llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/teamcraft-bench/teamcraft/blob/HEAD/llava_teamcraft/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2412.04449","paper":"/paper/p-mod-building-mixture-of-depths-mllms-via","title":"p-MoD: Building Mixture-of-Depths MLLMs via Progressive Ratio Decay","date":"2024-12-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mcg-nju/p-mod","path":"llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/mcg-nju/p-mod/blob/HEAD/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2412.04317","paper":"/paper/flashsloth-lightning-multimodal-large","title":"FlashSloth: Lightning Multimodal Large Language Models via Embedded Visual Compression","date":"2024-12-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"codefanw/flashsloth","path":"flashsloth/eval/model_vqa_mmbench.py","file_url":"https://github.com/codefanw/flashsloth/blob/HEAD/flashsloth/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2412.01818","paper":"/paper/cls-attention-is-all-you-need-for-training","title":"Beyond Text-Visual Attention: Exploiting Visual Cues for Effective Token Pruning in VLMs","date":"2024-12-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"theia-4869/fastervlm","path":"llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/theia-4869/fastervlm/blob/HEAD/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2411.16863","paper":"/paper/augmenting-multimodal-llms-with-self","title":"Augmenting Multimodal LLMs with Self-Reflective Tokens for Knowledge-based Visual Question Answering","date":"2024-11-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"aimagelab/reflectiva","path":"llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/aimagelab/reflectiva/blob/HEAD/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2411.16721","paper":"/paper/steering-away-from-harm-an-adaptive-approach","title":"Steering Away from Harm: An Adaptive Approach to Defending Vision Language Model Against Jailbreaks","date":"2024-11-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ASTRAL-Group/ASTRA","path":"utility_eval/minigpt_mmbench.py","file_url":"https://github.com/ASTRAL-Group/ASTRA/blob/HEAD/utility_eval/minigpt_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2411.10803","paper":"/paper/multi-stage-vision-token-dropping-towards","title":"Multi-Stage Vision Token Dropping: Towards Efficient Multimodal Large Language Model","date":"2024-11-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"liuting20/mustdrop","path":"llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/liuting20/mustdrop/blob/HEAD/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2411.02712","paper":"/paper/v-dpo-mitigating-hallucination-in-large","title":"V-DPO: Mitigating Hallucination in Large Vision Language Models via Vision-Guided Direct Preference Optimization","date":"2024-11-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yuxixie/v-dpo","path":"llava_dpo/eval/model_vqa_mmbench.py","file_url":"https://github.com/yuxixie/v-dpo/blob/HEAD/llava_dpo/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2410.17247","paper":"/paper/pyramiddrop-accelerating-your-large-vision","title":"PyramidDrop: Accelerating Your Large Vision-Language Models via Pyramid Visual Redundancy Reduction","date":"2024-10-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cooperx521/pyramiddrop","path":"llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/cooperx521/pyramiddrop/blob/HEAD/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2410.16236","paper":"/paper/llava-kd-a-framework-of-distilling-multimodal","title":"LLaVA-KD: A Framework of Distilling Multimodal Large Language Models","date":"2024-10-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Fantasyele/LLaVA-KD","path":"llavakd/eval/model_vqa_mmbench.py","file_url":"https://github.com/Fantasyele/LLaVA-KD/blob/HEAD/llavakd/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2410.16198","paper":"/paper/improve-vision-language-model-chain-of","title":"Improve Vision Language Model Chain-of-thought Reasoning","date":"2024-10-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"riflezhang/llava-reasoner-dpo","path":"llava_reasoner/llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/riflezhang/llava-reasoner-dpo/blob/HEAD/llava_reasoner/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2410.07167","paper":"/paper/deciphering-cross-modal-alignment-in-large","title":"Deciphering Cross-Modal Alignment in Large Vision-Language Models with Modality Integration Rate","date":"2024-10-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shikiw/modality-integration-rate","path":"llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/shikiw/modality-integration-rate/blob/HEAD/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2410.07113","paper":"/paper/personalized-visual-instruction-tuning","title":"Personalized Visual Instruction Tuning","date":"2024-10-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sterzhang/pvit","path":"personalize-llava/llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/sterzhang/pvit/blob/HEAD/personalize-llava/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2410.06625","paper":"/paper/eta-evaluating-then-aligning-safety-of-vision","title":"ETA: Evaluating Then Aligning Safety of Vision Language Models at Inference Time","date":"2024-10-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dripnowhy/eta","path":"llava/eval/model_vqa_mmbench_eta.py","file_url":"https://github.com/dripnowhy/eta/blob/HEAD/llava/eval/model_vqa_mmbench_eta.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2410.06169","paper":"/paper/quadratic-is-not-what-you-need-for-multimodal","title":"Treat Visual Tokens as Text? But Your MLLM Only Needs Fewer Efforts to See","date":"2024-10-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ZhangAIPI/YOPO_MLLM_Pruning","path":"LLaVA/llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/ZhangAIPI/YOPO_MLLM_Pruning/blob/HEAD/LLaVA/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2410.04417","paper":"/paper/sparsevlm-visual-token-sparsification-for","title":"SparseVLM: Visual Token Sparsification for Efficient Vision-Language Model Inference","date":"2024-10-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Gumpest/SparseVLMs","path":"llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/Gumpest/SparseVLMs/blob/HEAD/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2410.02080","paper":"/paper/emma-efficient-visual-alignment-in-multi","title":"EMMA: Efficient Visual Alignment in Multi-Modal LLMs","date":"2024-10-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"saraghazanfari/emma","path":"llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/saraghazanfari/emma/blob/HEAD/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2410.02745","paper":"/paper/avg-llava-a-large-multimodal-model-with","title":"AVG-LLaVA: A Large Multimodal Model with Adaptive Visual Granularity","date":"2024-09-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"deeplearnxmu/avg-llava","path":"llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/deeplearnxmu/avg-llava/blob/HEAD/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2409.17663","paper":"/paper/explanation-bottleneck-models","title":"Explanation Bottleneck Models","date":"2024-09-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yshinya6/xbm","path":"xbm-llava/llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/yshinya6/xbm/blob/HEAD/xbm-llava/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2409.10994","paper":"/paper/less-is-more-a-simple-yet-effective-token","title":"Less is More: A Simple yet Effective Token Reduction Method for Efficient Multi-modal LLMs","date":"2024-09-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"freedomintelligence/trim","path":"llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/freedomintelligence/trim/blob/HEAD/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2409.10197","paper":"/paper/fit-and-prune-fast-and-training-free-visual","title":"Fit and Prune: Fast and Training-free Visual Token Pruning for Multi-modal Large Language Models","date":"2024-09-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ywh187/fitprune","path":"LLaVA_1.5/llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/ywh187/fitprune/blob/HEAD/LLaVA_1.5/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2409.01179","paper":"/paper/recoverable-compression-a-multimodal-vision","title":"Recoverable Compression: A Multimodal Vision Token Recovery Mechanism Guided by Text Information","date":"2024-09-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"banjiuyufen/Recoverable-Compression","path":"llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/banjiuyufen/Recoverable-Compression/blob/HEAD/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2408.08862","paper":"/paper/visual-agents-as-fast-and-slow-thinkers","title":"Visual Agents as Fast and Slow Thinkers","date":"2024-08-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"guangyans/sys2-llava","path":"ROILLaVA/llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/guangyans/sys2-llava/blob/HEAD/ROILLaVA/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2408.03735","paper":"/paper/advancing-multimodal-large-language-models","title":"Advancing Multimodal Large Language Models with Quantization-Aware Scale Learning for Efficient Adaptation","date":"2024-08-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xjjxmu/qslaw","path":"llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/xjjxmu/qslaw/blob/HEAD/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2407.21771","paper":"/paper/paying-more-attention-to-image-a-training","title":"Paying More Attention to Image: A Training-Free Method for Alleviating Hallucination in LVLMs","date":"2024-07-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hasanar1f/llava-hallunication-fix","path":"modPAI/llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/hasanar1f/llava-hallunication-fix/blob/HEAD/modPAI/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2406.20098","paper":"/paper/web2code-a-large-scale-webpage-to-code","title":"Web2Code: A Large-scale Webpage-to-Code Dataset and Evaluation Framework for Multimodal LLMs","date":"2024-06-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"MBZUAI-LLM/web2code","path":"web2code/llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/MBZUAI-LLM/web2code/blob/HEAD/web2code/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2406.20092","paper":"/paper/llavolta-efficient-multi-modal-models-via","title":"Efficient Large Multi-modal Models via Visual Context Compression","date":"2024-06-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Beckschen/LLaVolta","path":"llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/Beckschen/LLaVolta/blob/HEAD/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2406.14852","paper":"/paper/is-a-picture-worth-a-thousand-words-delving","title":"Is A Picture Worth A Thousand Words? Delving Into Spatial Reasoning for Vision Language Models","date":"2024-06-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"BAAI-DCAI/Bunny","path":"bunny/eval/model_vqa_mmbench.py","file_url":"https://github.com/BAAI-DCAI/Bunny/blob/HEAD/bunny/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2406.14498","paper":"/paper/llasa-large-multimodal-agent-for-human","title":"LLaSA: A Multimodal LLM for Human Activity Analysis Through Wearable and Smartphone Sensors","date":"2024-06-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bashlab/llasa","path":"LLaSA/llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/bashlab/llasa/blob/HEAD/LLaSA/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2406.13642","paper":"/paper/spatialbot-precise-spatial-understanding-with","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","date":"2024-06-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"baai-dcai/spatialbot","path":"bunny/eval/model_vqa_mmbench.py","file_url":"https://github.com/baai-dcai/spatialbot/blob/HEAD/bunny/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2406.12275","paper":"/paper/voco-llama-towards-vision-compression-with","title":"VoCo-LLaMA: Towards Vision Compression with Large Language Models","date":"2024-06-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Yxxxb/VoCo-LLaMA","path":"llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/Yxxxb/VoCo-LLaMA/blob/HEAD/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2406.11280","paper":"/paper/i-srt-aligning-large-multimodal-models-for","title":"ISR-DPO: Aligning Large Multimodal Models for Videos by Iterative Self-Retrospective DPO","date":"2024-06-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"snumprlab/SRT","path":"llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/snumprlab/SRT/blob/HEAD/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2406.09400","paper":"/paper/yo-llava-your-personalized-language-and","title":"Yo'LLaVA: Your Personalized Language and Vision Assistant","date":"2024-06-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"WisconsinAIVision/YoLLaVA","path":"llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/WisconsinAIVision/YoLLaVA/blob/HEAD/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2406.08487","paper":"/paper/beyond-llava-hd-diving-into-high-resolution","title":"Beyond LLaVA-HD: Diving into High-Resolution Large Multimodal Models","date":"2024-06-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yfzhang114/slime","path":"llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/yfzhang114/slime/blob/HEAD/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2406.08100","paper":"/paper/multimodal-table-understanding","title":"Multimodal Table Understanding","date":"2024-06-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"spursgozmy/table-llava","path":"llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/spursgozmy/table-llava/blob/HEAD/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2406.02884","paper":"/paper/posterllava-constructing-a-unified-multi","title":"PosterLLaVa: Constructing a Unified Multi-modal Layout Generator with LLM","date":"2024-06-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"posterllava/posterllava","path":"llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/posterllava/posterllava/blob/HEAD/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2405.19315","paper":"/paper/matryoshka-query-transformer-for-large-vision","title":"Matryoshka Query Transformer for Large Vision-Language Models","date":"2024-05-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"gordonhu608/mqt-llava","path":"llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/gordonhu608/mqt-llava/blob/HEAD/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2405.15973","paper":"/paper/enhancing-visual-language-modality-alignment","title":"Enhancing Visual-Language Modality Alignment in Large Vision Language Models via Self-Improvement","date":"2024-05-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"umd-huang-lab/sima","path":"llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/umd-huang-lab/sima/blob/HEAD/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2405.14974","paper":"/paper/lova3-learning-to-visual-question-answering","title":"LOVA3: Learning to Visual Question Answering, Asking and Assessment","date":"2024-05-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"showlab/LOVA3","path":"llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/showlab/LOVA3/blob/HEAD/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2405.12107","paper":"/paper/imp-highly-capable-large-multimodal-models","title":"Imp: Highly Capable Large Multimodal Models for Mobile Devices","date":"2024-05-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"milvlg/imp","path":"imp_llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/milvlg/imp/blob/HEAD/imp_llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2405.05949","paper":"/paper/cumo-scaling-multimodal-llm-with-co-upcycled","title":"CuMo: Scaling Multimodal LLM with Co-Upcycled Mixture-of-Experts","date":"2024-05-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shi-labs/cumo","path":"cumo/eval/model_vqa_mmbench.py","file_url":"https://github.com/shi-labs/cumo/blob/HEAD/cumo/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2404.10501","paper":"/paper/self-supervised-visual-preference-alignment","title":"Self-Supervised Visual Preference Alignment","date":"2024-04-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Kevinz-code/SeVa","path":"seva/llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/Kevinz-code/SeVa/blob/HEAD/seva/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"GPL-3.0","inline_ok":false,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2404.05892","paper":"/paper/eagle-and-finch-rwkv-with-matrix-valued","title":"Eagle and Finch: RWKV with Matrix-Valued States and Dynamic Recurrence","date":"2024-04-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"howard-hou/visualrwkv","path":"VisualRWKV-v7/v7.10/evaluate_imagenet.py","file_url":"https://github.com/howard-hou/visualrwkv/blob/HEAD/VisualRWKV-v7/v7.10/evaluate_imagenet.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2403.20331","paper":"/paper/unsolvable-problem-detection-evaluating","title":"Unsolvable Problem Detection: Evaluating Trustworthiness of Vision Language Models","date":"2024-03-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"atsumiyai/upd","path":"automatic_eval/calculate_scores.py","file_url":"https://github.com/atsumiyai/upd/blob/HEAD/automatic_eval/calculate_scores.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0e7cf115754e1730","mcp_get_code":{"code_sha256":"0e7cf115754e1730"}},{"arxiv_id":"2403.18814","paper":"/paper/mini-gemini-mining-the-potential-of-multi","title":"Mini-Gemini: Mining the Potential of Multi-modality Vision Language Models","date":"2024-03-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dvlab-research/minigemini","path":"mgm/eval/model_vqa_mmbench.py","file_url":"https://github.com/dvlab-research/minigemini/blob/HEAD/mgm/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2403.18252","paper":"/paper/beyond-embeddings-the-promise-of-visual-table","title":"Beyond Embeddings: The Promise of Visual Table in Visual Reasoning","date":"2024-03-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lavi-lab/visual-table","path":"llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/lavi-lab/visual-table/blob/HEAD/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2403.16999","paper":"/paper/visual-cot-unleashing-chain-of-thought","title":"Visual CoT: Advancing Multi-Modal Language Models with a Comprehensive Dataset and Benchmark for Chain-of-Thought Reasoning","date":"2024-03-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"deepcs233/visual-cot","path":"llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/deepcs233/visual-cot/blob/HEAD/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2403.12966","paper":"/paper/chain-of-spot-interactive-reasoning-improves","title":"Chain-of-Spot: Interactive Reasoning Improves Large Vision-Language Models","date":"2024-03-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dongyh20/chain-of-spot","path":"llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/dongyh20/chain-of-spot/blob/HEAD/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2402.14545","paper":"/paper/less-is-more-mitigating-multimodal","title":"Less is More: Mitigating Multimodal Hallucination from an EOS Decision Perspective","date":"2024-02-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yuezih/less-is-more","path":"LLaVA/llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/yuezih/less-is-more/blob/HEAD/LLaVA/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2402.14418","paper":"/paper/uncertainty-aware-evaluation-for-vision","title":"Uncertainty-Aware Evaluation for Vision-Language Models","date":"2024-02-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ensec-ai/vlm-uncertainty-bench","path":"data_utils/common_utils.py","file_url":"https://github.com/ensec-ai/vlm-uncertainty-bench/blob/HEAD/data_utils/common_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"25593cfba3f7b483","mcp_get_code":{"code_sha256":"25593cfba3f7b483"}},{"arxiv_id":"2402.11530","paper":"/paper/efficient-multimodal-learning-from-data","title":"Efficient Multimodal Learning from Data-centric Perspective","date":"2024-02-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"baai-dcai/bunny","path":"bunny/eval/model_vqa_mmbench.py","file_url":"https://github.com/baai-dcai/bunny/blob/HEAD/bunny/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2402.11411","paper":"/paper/aligning-modalities-in-vision-large-language","title":"Aligning Modalities in Vision Large Language Models via Preference Fine-tuning","date":"2024-02-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yiyangzhou/povid","path":"llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/yiyangzhou/povid/blob/HEAD/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2402.10884","paper":"/paper/multi-modal-preference-alignment-remedies","title":"Multi-modal Preference Alignment Remedies Degradation of Visual Instruction Tuning on Language Models","date":"2024-02-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"findalexli/mllm-dpo","path":"llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/findalexli/mllm-dpo/blob/HEAD/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2402.03766","paper":"/paper/2402-03766","title":"MobileVLM V2: Faster and Stronger Baseline for Vision Language Model","date":"2024-02-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"meituan-automl/mobilevlm","path":"mobilevlm/eval/model_vqa_mmbench.py","file_url":"https://github.com/meituan-automl/mobilevlm/blob/HEAD/mobilevlm/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2401.06591","paper":"/paper/prometheus-vision-vision-language-model-as-a","title":"Prometheus-Vision: Vision-Language Model as a Judge for Fine-Grained Evaluation","date":"2024-01-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kaistai/prometheus-vision","path":"llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/kaistai/prometheus-vision/blob/HEAD/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2401.06209","paper":"/paper/eyes-wide-shut-exploring-the-visual","title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","date":"2024-01-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tsb0601/MMVP","path":"LLaVA/llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/tsb0601/MMVP/blob/HEAD/LLaVA/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2401.02330","paper":"/paper/llava-ph-efficient-multi-modal-assistant-with","title":"LLaVA-Phi: Efficient Multi-Modal Assistant with Small Language Model","date":"2024-01-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zhuyiche/llava-phi","path":"llava_phi/eval/model_vqa_mmbench.py","file_url":"https://github.com/zhuyiche/llava-phi/blob/HEAD/llava_phi/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2312.14233","paper":"/paper/vcoder-versatile-vision-encoders-for","title":"VCoder: Versatile Vision Encoders for Multimodal Large Language Models","date":"2023-12-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shi-labs/vcoder","path":"vcoder_llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/shi-labs/vcoder/blob/HEAD/vcoder_llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2312.06731","paper":"/paper/genixer-empowering-multimodal-large-language","title":"Genixer: Empowering Multimodal Large Language Models as a Powerful Data Generator","date":"2023-12-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zhaohengyuan1/genixer","path":"Genixer_LLaVA/llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/zhaohengyuan1/genixer/blob/HEAD/Genixer_LLaVA/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2311.17076","paper":"/paper/compositional-chain-of-thought-prompting-for","title":"Compositional Chain-of-Thought Prompting for Large Multimodal Models","date":"2023-11-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"chancharikmitra/ccot","path":"InstructBLIP-13b/InstructBLIP_MMBench.py","file_url":"https://github.com/chancharikmitra/ccot/blob/HEAD/InstructBLIP-13b/InstructBLIP_MMBench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2311.17043","paper":"/paper/llama-vid-an-image-is-worth-2-tokens-in-large","title":"LLaMA-VID: An Image is Worth 2 Tokens in Large Language Models","date":"2023-11-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dvlab-research/llama-vid","path":"llamavid/eval/model_vqa_mmbench.py","file_url":"https://github.com/dvlab-research/llama-vid/blob/HEAD/llamavid/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2311.07362","paper":"/paper/volcano-mitigating-multimodal-hallucination","title":"Volcano: Mitigating Multimodal Hallucination through Self-Feedback Guided Revision","date":"2023-11-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kaistai/volcano","path":"llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/kaistai/volcano/blob/HEAD/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2310.08588","paper":"/paper/octopus-embodied-vision-language-programmer","title":"Octopus: Embodied Vision-Language Programmer from Environmental Feedback","date":"2023-10-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dongyh20/octopus","path":"octopus/LLaVA/llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/dongyh20/octopus/blob/HEAD/octopus/LLaVA/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2308.13566","paper":"/paper/mllm-dataengine-an-iterative-refinement","title":"MLLM-DataEngine: An Iterative Refinement Approach for MLLM","date":"2023-08-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"opendatalab/mllm-dataengine","path":"LLaVA/llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/opendatalab/mllm-dataengine/blob/HEAD/LLaVA/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2308.10253","paper":"/paper/stablellava-enhanced-visual-instruction","title":"StableLLaVA: Enhanced Visual Instruction Tuning with Synthesized Image-Dialogue Data","date":"2023-08-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"icoz69/stablellava","path":"llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/icoz69/stablellava/blob/HEAD/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2112.05682","paper":"/paper/self-attention-does-not-need-o-n-2-memory","title":"Self-attention Does Not Need $O(n^2)$ Memory","date":"2021-12-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jihaonew/mm-instruct","path":"llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/jihaonew/mm-instruct/blob/HEAD/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"Xing_Conical_Visual_Concentration_for_Efficient_Large_Vision-Language_Models_CVPR_2025_paper","paper":null,"title":"arXiv:Xing_Conical_Visual_Concentration_for_Efficient_Large_Vision-Language_Models_CVPR_2025_paper","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"Cooperx521/PyramidDrop","path":"llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/Cooperx521/PyramidDrop/blob/HEAD/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2025.findings-emnlp.1095","paper":null,"title":"arXiv:2025.findings-emnlp.1095","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"AngelAlita/AsD","path":"llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/AngelAlita/AsD/blob/HEAD/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2025.findings-acl.865","paper":null,"title":"arXiv:2025.findings-acl.865","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"DeepLearnXMU/AVG-LLaVA","path":"llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/DeepLearnXMU/AVG-LLaVA/blob/HEAD/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2025.findings-acl.327","paper":null,"title":"arXiv:2025.findings-acl.327","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"SakuraTroyChen/PyPE","path":"LLaVA/llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/SakuraTroyChen/PyPE/blob/HEAD/LLaVA/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2024.findings-naacl.226","paper":null,"title":"arXiv:2024.findings-naacl.226","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"nguyennm1024/OSCaR","path":"llava/eval/model_vqa_mmbench.py","file_url":"https://github.com/nguyennm1024/OSCaR/blob/HEAD/llava/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}},{"arxiv_id":"2024.findings-emnlp.775","paper":null,"title":"arXiv:2024.findings-emnlp.775","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"YuxiXie/V-DPO","path":"llava_dpo/eval/model_vqa_mmbench.py","file_url":"https://github.com/YuxiXie/V-DPO/blob/HEAD/llava_dpo/eval/model_vqa_mmbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bae18947b56f2be1","mcp_get_code":{"code_sha256":"bae18947b56f2be1"}}]}