{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/get-chunk","entry":"get_chunk","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":197,"n_papers_ran":193,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":16,"n_samples_ran":9,"n_samples_fingerprinted":8,"n_places":203,"n_places_pointer_only":66,"by_status":{"ran_honours":1,"ran_violates":0,"ran_draft_wrong":5,"ran_fixture":0,"ran":3,"unverified":7},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2609.00231","paper":"/paper/arxiv-2609-00231","title":"Beyond Language Priors: Diagnosing and Fixing Visual-Origin Hallucinations in Multimodal LLM","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"zxp555/ACFT_MM26","path":"ACFT/llava/eval/model_vqa.py","file_url":"https://github.com/zxp555/ACFT_MM26/blob/HEAD/ACFT/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2608.17070","paper":"/paper/arxiv-2608-17070","title":"Certified but Private: Scalable Zero-Knowledge Proofs for Neural Network Guarantees","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"youweizhong/PANDA","path":"evaluation/config.py","file_url":"https://github.com/youweizhong/PANDA/blob/HEAD/evaluation/config.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c39616a6f06d7a49","mcp_get_code":{"code_sha256":"c39616a6f06d7a49"}},{"arxiv_id":"2608.04496","paper":"/paper/arxiv-2608-04496","title":"DIVE: Dynamic Iterative Visual Evidence Construction for Efficient Vision-Language Models","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"Zhong-Chenchen/DIVE","path":"llava/eval/model_vqa.py","file_url":"https://github.com/Zhong-Chenchen/DIVE/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2607.15299","paper":"/paper/arxiv-2607-15299","title":"MLLM-DataEngine: Closing the Loop of Multimodal Instruction Tuning Data Generation","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"opendatalab/MLLM-DataEngine","path":"LLaVA/llava/eval/model_vqa.py","file_url":"https://github.com/opendatalab/MLLM-DataEngine/blob/HEAD/LLaVA/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2605.27378","paper":"/paper/arxiv-2605-27378","title":"OralAgent: Integrating Reasoning, Tools, and Knowledge for Interactive Dental Image Analysis","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"isjinghao/OralAgent","path":"oralagent/llava/eval/model_vqa.py","file_url":"https://github.com/isjinghao/OralAgent/blob/HEAD/oralagent/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2604.02570","paper":"/paper/arxiv-2604-02570","title":"WSVD: Weighted Low-Rank Approximation for Fast and Efficient Execution of Low-Precision Vision-Language Models","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"SAI-Lab-NYU/WSVD","path":"e2e/infer_llava.py","file_url":"https://github.com/SAI-Lab-NYU/WSVD/blob/HEAD/e2e/infer_llava.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2603.21426","paper":"/paper/arxiv-2603-21426","title":"Uncertainty-Aware Knowledge Distillation for Multimodal Large Language Models","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"Jingchensun/beta-kd","path":"mobilevlm/eval/model_vqa_loader.py","file_url":"https://github.com/Jingchensun/beta-kd/blob/HEAD/mobilevlm/eval/model_vqa_loader.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2602.23699","paper":"/paper/arxiv-2602-23699","title":"HiDrop: Hierarchical Vision Token Reduction in MLLMs via Late Injection, Concave Pyramid Pruning, and Early Exit","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"EIT-NLP/HiDrop","path":"llava/eval/model_vqa.py","file_url":"https://github.com/EIT-NLP/HiDrop/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2602.09483","paper":"/paper/arxiv-2602-09483","title":"Beyond Next-Token Alignment: Distilling Multimodal Large Language Models via Token Interactions","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"lchen1019/Align-TI","path":"alignti/eval/model_vqa.py","file_url":"https://github.com/lchen1019/Align-TI/blob/HEAD/alignti/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2601.17918","paper":"/paper/arxiv-2601-17918","title":"Benchmarking Direct Preference Optimization for Medical Large Vision-Language Models","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"dmis-lab/med-vlm-dpo","path":"inference/LLaVA-Med/llava/eval/model_vqa.py","file_url":"https://github.com/dmis-lab/med-vlm-dpo/blob/HEAD/inference/LLaVA-Med/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2601.14732","paper":"/paper/arxiv-2601-14732","title":"DeepMoLM: Leveraging Visual and Geometric Structural Information for Molecule-Text Modeling","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"1anj/DeepMoLM","path":"llava/eval/model_vqa_video.py","file_url":"https://github.com/1anj/DeepMoLM/blob/HEAD/llava/eval/model_vqa_video.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2601.10710","paper":"/paper/arxiv-2601-10710","title":"Cross-Layer Injection for Deep Vision-Language Fusion","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"codefuse-ai/CLI","path":"llava/eval/model_vqa.py","file_url":"https://github.com/codefuse-ai/CLI/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2510.16292","paper":"/paper/arxiv-2510-16292","title":"QSVD: Efficient Low-rank Approximation for Unified Query-Key-Value Weight Compression in Low-Precision Vision-Language Models","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"SAI-Lab-NYU/QSVD","path":"fake_quant/eval_llavanext_vizwiz.py","file_url":"https://github.com/SAI-Lab-NYU/QSVD/blob/HEAD/fake_quant/eval_llavanext_vizwiz.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2510.14374","paper":"/paper/arxiv-2510-14374","title":"Spatial Preference Rewarding for MLLMs Spatial Understanding","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"hanqiu-hq/SPR","path":"construct_data/ferret_score_siglip.py","file_url":"https://github.com/hanqiu-hq/SPR/blob/HEAD/construct_data/ferret_score_siglip.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2507.18300","paper":null,"title":"arXiv:2507.18300","date":null,"month_inferred_from_arxiv_id":"2025-07","title_source":null,"repo":"360CVGroup/LMM-Det","path":"llava/eval/model_coco_owlv2.py","file_url":"https://github.com/360CVGroup/LMM-Det/blob/HEAD/llava/eval/model_coco_owlv2.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2507.00505","paper":"/paper/llava-sp-enhancing-visual-representation-with","title":"LLaVA-SP: Enhancing Visual Representation with Visual Spatial Tokens for MLLMs","date":"2025-07-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"CnFaker/LLaVA-SP","path":"llava/eval/model_vqa.py","file_url":"https://github.com/CnFaker/LLaVA-SP/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2506.17202","paper":"/paper/unifork-exploring-modality-alignment-for","title":"UniFork: Exploring Modality Alignment for Unified Multimodal Understanding and Generation","date":"2025-06-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tliby/unifork","path":"unifork/eval/model_vqa.py","file_url":"https://github.com/tliby/unifork/blob/HEAD/unifork/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2506.10967","paper":"/paper/beyond-attention-or-similarity-maximizing","title":"Beyond Attention or Similarity: Maximizing Conditional Diversity for Token Pruning in MLLMs","date":"2025-06-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"theia-4869/cdpruner","path":"llava/eval/model_vqa.py","file_url":"https://github.com/theia-4869/cdpruner/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2505.24625","paper":"/paper/learning-from-videos-for-3d-world-enhancing","title":"Learning from Videos for 3D World: Enhancing MLLMs with 3D Vision Geometry Priors","date":"2025-05-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"LaVi-Lab/Video-3D-LLM","path":"llava/eval/model_vqa.py","file_url":"https://github.com/LaVi-Lab/Video-3D-LLM/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2505.23359","paper":"/paper/videoreasonbench-can-mllms-perform-vision","title":"VideoReasonBench: Can MLLMs Perform Vision-Centric Complex Video Reasoning?","date":"2025-05-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"llyx97/video_reason_bench","path":"eval_api.py","file_url":"https://github.com/llyx97/video_reason_bench/blob/HEAD/eval_api.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"cda3e7eebb6da37a","mcp_get_code":{"code_sha256":"cda3e7eebb6da37a"}},{"arxiv_id":"2505.20753","paper":"/paper/understand-think-and-answer-advancing-visual","title":"Understand, Think, and Answer: Advancing Visual Reasoning with Large Multimodal Models","date":"2025-05-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jefferyzhan/griffon","path":"griffon/eval/model_vqa.py","file_url":"https://github.com/jefferyzhan/griffon/blob/HEAD/griffon/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2505.16839","paper":"/paper/lavida-a-large-diffusion-language-model-for","title":"LaViDa: A Large Diffusion Language Model for Multimodal Understanding","date":"2025-05-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jacklishufan/lavida","path":"llava/eval/model_vqa.py","file_url":"https://github.com/jacklishufan/lavida/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2505.02835","paper":"/paper/r1-reward-training-multimodal-reward-model","title":"R1-Reward: Training Multimodal Reward Model Through Stable Reinforcement Learning","date":"2025-05-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yfzhang114/r1_reward","path":"inference/MM-RLHF-Reward/r1_reward.py","file_url":"https://github.com/yfzhang114/r1_reward/blob/HEAD/inference/MM-RLHF-Reward/r1_reward.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2505.01456","paper":"/paper/unlearning-sensitive-information-in","title":"Unlearning Sensitive Information in Multimodal LLMs: Benchmark and Attack-Defense Evaluation","date":"2025-05-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"vaidehi99/unlok-vqa","path":"LLaVA/llava/eval/model_vqa.py","file_url":"https://github.com/vaidehi99/unlok-vqa/blob/HEAD/LLaVA/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2504.18856","paper":"/paper/multi-resolution-pathology-language-pre-1","title":"Multi-Resolution Pathology-Language Pre-training Model with Text-Guided Visual Representation","date":"2025-04-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"BasitAlawode/MR-PLIP","path":"generate_text.py","file_url":"https://github.com/BasitAlawode/MR-PLIP/blob/HEAD/generate_text.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2504.14988","paper":"/paper/benchmarking-large-vision-language-models-on","title":"Benchmarking Large Vision-Language Models on Fine-Grained Image Tasks: A Comprehensive Evaluation","date":"2025-04-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"SEU-VIPGroup/FG-BMK","path":"demo/human_evaluation/human_evaluation_demo.py","file_url":"https://github.com/SEU-VIPGroup/FG-BMK/blob/HEAD/demo/human_evaluation/human_evaluation_demo.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"49a93847dfeff479","mcp_get_code":{"code_sha256":"49a93847dfeff479"}},{"arxiv_id":"2504.07934","paper":"/paper/sota-with-less-mcts-guided-sample-selection","title":"SoTA with Less: MCTS-Guided Sample Selection for Data-Efficient Visual Reasoning Self-Improvement","date":"2025-04-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2504.07934","paper":"/paper/sota-with-less-mcts-guided-sample-selection","title":"SoTA with Less: MCTS-Guided Sample Selection for Data-Efficient Visual Reasoning Self-Improvement","date":"2025-04-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"si0wang/thinklite-vl","path":"eval/model_ai2d_qwen.py","file_url":"https://github.com/si0wang/thinklite-vl/blob/HEAD/eval/model_ai2d_qwen.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e6e072225bd8f400","mcp_get_code":{"code_sha256":"e6e072225bd8f400"}},{"arxiv_id":"2504.00502","paper":null,"title":"arXiv:2504.00502","date":null,"month_inferred_from_arxiv_id":"2025-04","title_source":null,"repo":"icip-cas/ShortV","path":"llava/eval/model_vqa.py","file_url":"https://github.com/icip-cas/ShortV/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2503.13440","paper":"/paper/matvlm-hybrid-mamba-transformer-for-efficient","title":"MaTVLM: Hybrid Mamba-Transformer for Efficient Vision-Language Modeling","date":"2025-03-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hustvl/MaTVLM","path":"tinyllava/eval/model_vqa.py","file_url":"https://github.com/hustvl/MaTVLM/blob/HEAD/tinyllava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2503.11832","paper":"/paper/safety-mirage-how-spurious-correlations","title":"Safety Mirage: How Spurious Correlations Undermine VLM Safety Fine-tuning","date":"2025-03-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"OPTML-Group/VLM-Safety-Unlearn","path":"llava/eval/model_vqa.py","file_url":"https://github.com/OPTML-Group/VLM-Safety-Unlearn/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2503.10742","paper":"/paper/keyframe-oriented-vision-token-pruning","title":"Keyframe-oriented Vision Token Pruning: Enhancing Efficiency of Large Vision Language Models on Long-Form Video Processing","date":"2025-03-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"1999Lyd/KVTP","path":"llava/eval/model_vqa.py","file_url":"https://github.com/1999Lyd/KVTP/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2503.10501","paper":"/paper/tokencarve-information-preserving-visual","title":"TokenCarve: Information-Preserving Visual Token Compression in Multimodal Large Language Models","date":"2025-03-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shawntan86/tokencarve","path":"TokenCarve/TokenCarve_model_vqa_loader.py","file_url":"https://github.com/shawntan86/tokencarve/blob/HEAD/TokenCarve/TokenCarve_model_vqa_loader.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2502.13146","paper":"/paper/re-align-aligning-vision-language-models-via","title":"Re-Align: Aligning Vision Language Models via Retrieval-Augmented Direct Preference Optimization","date":"2025-02-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"taco-group/re-align","path":"llava/eval/model_vqa.py","file_url":"https://github.com/taco-group/re-align/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2502.11494","paper":"/paper/stop-looking-for-important-tokens-in","title":"Stop Looking for Important Tokens in Multimodal Language Models: Duplication Matters More","date":"2025-02-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zichenwen1/dart","path":"llava/eval/model_vqa.py","file_url":"https://github.com/zichenwen1/dart/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2502.09838","paper":"/paper/healthgpt-a-medical-large-vision-language","title":"HealthGPT: A Medical Large Vision-Language Model for Unifying Comprehension and Generation via Heterogeneous Knowledge Adaptation","date":"2025-02-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dcdmllm/healthgpt","path":"HealthGPT/llava/eval/model_vqa.py","file_url":"https://github.com/dcdmllm/healthgpt/blob/HEAD/HealthGPT/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2502.06788","paper":"/paper/evev2-improved-baselines-for-encoder-free","title":"EVEv2: Improved Baselines for Encoder-Free Vision-Language Models","date":"2025-02-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"baaivision/EVE","path":"EVEv1/eve/eval/model_vqa.py","file_url":"https://github.com/baaivision/EVE/blob/HEAD/EVEv1/eve/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2502.05173","paper":"/paper/videorope-what-makes-for-good-video-rotary","title":"VideoRoPE: What Makes for Good Video Rotary Position Embedding?","date":"2025-02-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wiselnn570/videorope","path":"eval/model_longvideobench_qwen2_vl.py","file_url":"https://github.com/wiselnn570/videorope/blob/HEAD/eval/model_longvideobench_qwen2_vl.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"e32e49fba862aa72","mcp_get_code":{"code_sha256":"e32e49fba862aa72"}},{"arxiv_id":"2502.05173","paper":"/paper/videorope-what-makes-for-good-video-rotary","title":"VideoRoPE: What Makes for Good Video Rotary Position Embedding?","date":"2025-02-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wiselnn570/videorope","path":"eval/model_videohallucer.py","file_url":"https://github.com/wiselnn570/videorope/blob/HEAD/eval/model_videohallucer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"b6f76fcc05062c30","mcp_get_code":{"code_sha256":"b6f76fcc05062c30"}},{"arxiv_id":"2502.02673","paper":"/paper/medrax-medical-reasoning-agent-for-chest-x","title":"MedRAX: Medical Reasoning Agent for Chest X-ray","date":"2025-02-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bowang-lab/medrax","path":"medrax/llava/eval/model_vqa.py","file_url":"https://github.com/bowang-lab/medrax/blob/HEAD/medrax/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2501.09695","paper":"/paper/mitigating-hallucinations-in-large-vision-3","title":"Mitigating Hallucinations in Large Vision-Language Models via DPO: On-Policy Data Hold the Key","date":"2025-01-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zhyang2226/opa-dpo","path":"eval_llava_rlhf_coco/model_vqa.py","file_url":"https://github.com/zhyang2226/opa-dpo/blob/HEAD/eval_llava_rlhf_coco/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2501.07888","paper":"/paper/tarsier2-advancing-large-vision-language","title":"Tarsier2: Advancing Large Vision-Language Models from Detailed Video Description to Comprehensive Video Understanding","date":"2025-01-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2412.12359","paper":"/paper/visual-instruction-tuning-with-500x-fewer","title":"LLaVA Steering: Visual Instruction Tuning with 500x Fewer Parameters through Modality Linear Representation-Steering","date":"2024-12-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bibisbar/LLaVA-Steering","path":"tinyllava/eval/model_vqa.py","file_url":"https://github.com/bibisbar/LLaVA-Steering/blob/HEAD/tinyllava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2412.12359","paper":"/paper/visual-instruction-tuning-with-500x-fewer","title":"LLaVA Steering: Visual Instruction Tuning with 500x Fewer Parameters through Modality Linear Representation-Steering","date":"2024-12-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bibisbar/LLaVA-Steering","path":"tinyllava/eval/model_vqa_chair.py","file_url":"https://github.com/bibisbar/LLaVA-Steering/blob/HEAD/tinyllava/eval/model_vqa_chair.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"01c6b696dad5f567","mcp_get_code":{"code_sha256":"01c6b696dad5f567"}},{"arxiv_id":"2412.09501","paper":"/paper/lyra-an-efficient-and-speech-centric","title":"Lyra: An Efficient and Speech-Centric Framework for Omni-Cognition","date":"2024-12-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dvlab-research/Lyra","path":"lyra/eval/model_lyra_image_speech.py","file_url":"https://github.com/dvlab-research/Lyra/blob/HEAD/lyra/eval/model_lyra_image_speech.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2412.07689","paper":"/paper/drivemm-all-in-one-large-multimodal-model-for","title":"DriveMM: All-in-One Large Multimodal Model for Autonomous Driving","date":"2024-12-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zhijian11/DriveMM","path":"llava/eval/model_vqa.py","file_url":"https://github.com/zhijian11/DriveMM/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2412.06141","paper":"/paper/mmedpo-aligning-medical-vision-language","title":"MMedPO: Aligning Medical Vision-Language Models with Clinical-Aware Multimodal Preference Optimization","date":"2024-12-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"aiming-lab/mmedpo","path":"inference/llava-med-1.5_report.py","file_url":"https://github.com/aiming-lab/mmedpo/blob/HEAD/inference/llava-med-1.5_report.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2412.05819","paper":"/paper/cls-token-tells-everything-needed-for","title":"[CLS] Token Tells Everything Needed for Training-free Efficient MLLMs","date":"2024-12-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thu-mig/vtc-cls","path":"llava/eval/model_vqa.py","file_url":"https://github.com/thu-mig/vtc-cls/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2412.05255","paper":"/paper/teamcraft-a-benchmark-for-multi-modal-multi","title":"TeamCraft: A Benchmark for Multi-Modal Multi-Agent Systems in Minecraft","date":"2024-12-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"teamcraft-bench/teamcraft","path":"llava_teamcraft/llava/eval/model_vqa.py","file_url":"https://github.com/teamcraft-bench/teamcraft/blob/HEAD/llava_teamcraft/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2412.04449","paper":"/paper/p-mod-building-mixture-of-depths-mllms-via","title":"p-MoD: Building Mixture-of-Depths MLLMs via Progressive Ratio Decay","date":"2024-12-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mcg-nju/p-mod","path":"llava/eval/model_vqa.py","file_url":"https://github.com/mcg-nju/p-mod/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2412.04317","paper":"/paper/flashsloth-lightning-multimodal-large","title":"FlashSloth: Lightning Multimodal Large Language Models via Embedded Visual Compression","date":"2024-12-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"codefanw/flashsloth","path":"flashsloth/eval/model_vqa.py","file_url":"https://github.com/codefanw/flashsloth/blob/HEAD/flashsloth/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2412.03248","paper":"/paper/aim-adaptive-inference-of-multi-modal-llms","title":"AIM: Adaptive Inference of Multi-Modal LLMs via Token Merging and Pruning","date":"2024-12-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lavi-lab/aim","path":"llava/eval/model_vqa.py","file_url":"https://github.com/lavi-lab/aim/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2412.02158","paper":"/paper/agri-llava-knowledge-infused-large-multimodal","title":"Agri-LLaVA: Knowledge-Infused Large Multimodal Assistant on Agricultural Pests and Diseases","date":"2024-12-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kki2eve/agri-llava","path":"agri_llava/eval/model_vqa.py","file_url":"https://github.com/kki2eve/agri-llava/blob/HEAD/agri_llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2412.01818","paper":"/paper/cls-attention-is-all-you-need-for-training","title":"Beyond Text-Visual Attention: Exploiting Visual Cues for Effective Token Pruning in VLMs","date":"2024-12-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"theia-4869/fastervlm","path":"llava/eval/model_vqa.py","file_url":"https://github.com/theia-4869/fastervlm/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2412.00876","paper":"/paper/dynamic-llava-efficient-multimodal-large","title":"Dynamic-LLaVA: Efficient Multimodal Large Language Models via Dynamic Vision-language Context Sparsification","date":"2024-12-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Osilly/dynamic_llava","path":"llava/dynamic_eval/model_lvis_for_meteor.py","file_url":"https://github.com/Osilly/dynamic_llava/blob/HEAD/llava/dynamic_eval/model_lvis_for_meteor.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2411.16863","paper":"/paper/augmenting-multimodal-llms-with-self","title":"Augmenting Multimodal LLMs with Self-Reflective Tokens for Knowledge-based Visual Question Answering","date":"2024-11-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"aimagelab/reflectiva","path":"llava/eval/model_vqa.py","file_url":"https://github.com/aimagelab/reflectiva/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2411.16724","paper":"/paper/devils-in-middle-layers-of-large-vision","title":"Devils in Middle Layers of Large Vision-Language Models: Interpreting, Detecting and Mitigating Object Hallucinations via Attention Lens","date":"2024-11-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zhangqijiang07/middle_layers_indicating_hallucinations","path":"llava/eval/model_vqa.py","file_url":"https://github.com/zhangqijiang07/middle_layers_indicating_hallucinations/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2411.16044","paper":"/paper/zoomeye-enhancing-multimodal-llms-with-human","title":"ZoomEye: Enhancing Multimodal LLMs with Human-Like Zooming Capabilities through Tree-Based Image Exploration","date":"2024-11-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"om-ai-lab/ZoomEye","path":"ZoomEye/eval/perform_zoom_eye.py","file_url":"https://github.com/om-ai-lab/ZoomEye/blob/HEAD/ZoomEye/eval/perform_zoom_eye.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d61cdd0662f59fed","mcp_get_code":{"code_sha256":"d61cdd0662f59fed"}},{"arxiv_id":"2411.15024","paper":"/paper/dycoke-dynamic-compression-of-tokens-for-fast","title":"DyCoke: Dynamic Compression of Tokens for Fast Video Large Language Models","date":"2024-11-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kd-tao/dycoke","path":"llava/eval/model_vqa.py","file_url":"https://github.com/kd-tao/dycoke/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2411.14432","paper":"/paper/insight-v-exploring-long-chain-visual","title":"Insight-V: Exploring Long-Chain Visual Reasoning with Multimodal Large Language Models","date":"2024-11-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dongyh20/insight-v","path":"llava/eval/model_vqa.py","file_url":"https://github.com/dongyh20/insight-v/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2411.11360","paper":"/paper/ccexpert-advancing-mllm-capability-in-remote","title":"CCExpert: Advancing MLLM Capability in Remote Sensing Change Captioning with Difference-Aware Integration and a Foundational Dataset","date":"2024-11-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"meize0729/ccexpert","path":"llava/eval/model_vqa.py","file_url":"https://github.com/meize0729/ccexpert/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2411.11066","paper":"/paper/ts-llava-constructing-visual-tokens-through","title":"TS-LLaVA: Constructing Visual Tokens through Thumbnail-and-Sampling for Training-Free Video Large Language Models","date":"2024-11-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tingyu215/ts-llava","path":"llava/eval/run_inference_benchmark_consistency.py","file_url":"https://github.com/tingyu215/ts-llava/blob/HEAD/llava/eval/run_inference_benchmark_consistency.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2411.10803","paper":"/paper/multi-stage-vision-token-dropping-towards","title":"Multi-Stage Vision Token Dropping: Towards Efficient Multimodal Large Language Model","date":"2024-11-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"liuting20/mustdrop","path":"llava/eval/model_vqa.py","file_url":"https://github.com/liuting20/mustdrop/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2411.02712","paper":"/paper/v-dpo-mitigating-hallucination-in-large","title":"V-DPO: Mitigating Hallucination in Large Vision Language Models via Vision-Guided Direct Preference Optimization","date":"2024-11-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yuxixie/v-dpo","path":"llava_dpo/eval/model_vqa.py","file_url":"https://github.com/yuxixie/v-dpo/blob/HEAD/llava_dpo/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2410.17885","paper":"/paper/r-cot-reverse-chain-of-thought-problem","title":"R-CoT: Reverse Chain-of-Thought Problem Generation for Geometric Reasoning in Large Multimodal Models","date":"2024-10-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dle666/r-cot","path":"GeoQA_test/model_vqa_rcot7b.py","file_url":"https://github.com/dle666/r-cot/blob/HEAD/GeoQA_test/model_vqa_rcot7b.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2410.17637","paper":"/paper/mia-dpo-multi-image-augmented-direct","title":"MIA-DPO: Multi-Image Augmented Direct Preference Optimization For Large Vision-Language Models","date":"2024-10-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"liuziyu77/mia-dpo","path":"LLaVA-Hound-DPO/chatuniv/ChatUniVi/eval/model_coco_vqa.py","file_url":"https://github.com/liuziyu77/mia-dpo/blob/HEAD/LLaVA-Hound-DPO/chatuniv/ChatUniVi/eval/model_coco_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2410.17247","paper":"/paper/pyramiddrop-accelerating-your-large-vision","title":"PyramidDrop: Accelerating Your Large Vision-Language Models via Pyramid Visual Redundancy Reduction","date":"2024-10-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cooperx521/pyramiddrop","path":"llava/eval/model_vqa.py","file_url":"https://github.com/cooperx521/pyramiddrop/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2410.16236","paper":"/paper/llava-kd-a-framework-of-distilling-multimodal","title":"LLaVA-KD: A Framework of Distilling Multimodal Large Language Models","date":"2024-10-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Fantasyele/LLaVA-KD","path":"llavakd/eval/model_vqa.py","file_url":"https://github.com/Fantasyele/LLaVA-KD/blob/HEAD/llavakd/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2410.16198","paper":"/paper/improve-vision-language-model-chain-of","title":"Improve Vision Language Model Chain-of-thought Reasoning","date":"2024-10-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"riflezhang/llava-hound-dpo","path":"llava_hound_dpo/inference/run_inference_video_caption.py","file_url":"https://github.com/riflezhang/llava-hound-dpo/blob/HEAD/llava_hound_dpo/inference/run_inference_video_caption.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2410.13085","paper":"/paper/mmed-rag-versatile-multimodal-rag-system-for","title":"MMed-RAG: Versatile Multimodal RAG System for Medical Vision Language Models","date":"2024-10-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"richard-peng-xia/MMed-RAG","path":"train/dpo/povid_infer.py","file_url":"https://github.com/richard-peng-xia/MMed-RAG/blob/HEAD/train/dpo/povid_infer.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2410.11842","paper":"/paper/moh-multi-head-attention-as-mixture-of-head","title":"MoH: Multi-Head Attention as Mixture-of-Head Attention","date":"2024-10-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pku-yuangroup/chat-univi","path":"ChatUniVi/eval/model_coco_vqa.py","file_url":"https://github.com/pku-yuangroup/chat-univi/blob/HEAD/ChatUniVi/eval/model_coco_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2410.09855","paper":"/paper/text4seg-reimagining-image-segmentation-as","title":"Text4Seg: Reimagining Image Segmentation as Text Generation","date":"2024-10-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mc-lan/Text4Seg","path":"ms-swift/text4seg/infer_refer_seg.py","file_url":"https://github.com/mc-lan/Text4Seg/blob/HEAD/ms-swift/text4seg/infer_refer_seg.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"fec9d7679b542a15","mcp_get_code":{"code_sha256":"fec9d7679b542a15"}},{"arxiv_id":"2410.08119","paper":"/paper/q-vlm-post-training-quantization-for-large","title":"Q-VLM: Post-training Quantization for Large Vision-Language Models","date":"2024-10-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"changyuanwang17/qvlm","path":"llava/eval/model_vqa.py","file_url":"https://github.com/changyuanwang17/qvlm/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2410.07167","paper":"/paper/deciphering-cross-modal-alignment-in-large","title":"Deciphering Cross-Modal Alignment in Large Vision-Language Models with Modality Integration Rate","date":"2024-10-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shikiw/modality-integration-rate","path":"llava/eval/model_vqa.py","file_url":"https://github.com/shikiw/modality-integration-rate/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2410.07113","paper":"/paper/personalized-visual-instruction-tuning","title":"Personalized Visual Instruction Tuning","date":"2024-10-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sterzhang/pvit","path":"personalize-llava/llava/eval/model_vqa.py","file_url":"https://github.com/sterzhang/pvit/blob/HEAD/personalize-llava/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2410.06625","paper":"/paper/eta-evaluating-then-aligning-safety-of-vision","title":"ETA: Evaluating Then Aligning Safety of Vision Language Models at Inference Time","date":"2024-10-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dripnowhy/eta","path":"llava/eval/model_vqa_loader_eta.py","file_url":"https://github.com/dripnowhy/eta/blob/HEAD/llava/eval/model_vqa_loader_eta.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2410.06169","paper":"/paper/quadratic-is-not-what-you-need-for-multimodal","title":"Treat Visual Tokens as Text? But Your MLLM Only Needs Fewer Efforts to See","date":"2024-10-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ZhangAIPI/YOPO_MLLM_Pruning","path":"LLaVA/llava/eval/model_vqa.py","file_url":"https://github.com/ZhangAIPI/YOPO_MLLM_Pruning/blob/HEAD/LLaVA/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2410.04417","paper":"/paper/sparsevlm-visual-token-sparsification-for","title":"SparseVLM: Visual Token Sparsification for Efficient Vision-Language Model Inference","date":"2024-10-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Gumpest/SparseVLMs","path":"llava/eval/model_vqa.py","file_url":"https://github.com/Gumpest/SparseVLMs/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2410.03577","paper":"/paper/look-twice-before-you-answer-memory-space","title":"Look Twice Before You Answer: Memory-Space Visual Retracing for Hallucination Mitigation in Multimodal Large Language Models","date":"2024-10-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"1zhou-Wang/MemVR","path":"eval/glm_model_vqa.py","file_url":"https://github.com/1zhou-Wang/MemVR/blob/HEAD/eval/glm_model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2410.02080","paper":"/paper/emma-efficient-visual-alignment-in-multi","title":"EMMA: Efficient Visual Alignment in Multi-Modal LLMs","date":"2024-10-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"saraghazanfari/emma","path":"llava/eval/model_vqa.py","file_url":"https://github.com/saraghazanfari/emma/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2410.02745","paper":"/paper/avg-llava-a-large-multimodal-model-with","title":"AVG-LLaVA: A Large Multimodal Model with Adaptive Visual Granularity","date":"2024-09-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"deeplearnxmu/avg-llava","path":"llava/eval/model_vqa.py","file_url":"https://github.com/deeplearnxmu/avg-llava/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2409.17663","paper":"/paper/explanation-bottleneck-models","title":"Explanation Bottleneck Models","date":"2024-09-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yshinya6/xbm","path":"xbm-llava/llava/eval/model_vqa.py","file_url":"https://github.com/yshinya6/xbm/blob/HEAD/xbm-llava/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2409.17647","paper":"/paper/mecd-unlocking-multi-event-causal-discovery","title":"MECD: Unlocking Multi-Event Causal Discovery in Video Reasoning","date":"2024-09-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2409.16597","paper":"/paper/eventhallusion-diagnosing-event","title":"EventHallusion: Diagnosing Event Hallucinations in Video LLMs","date":"2024-09-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2409.16261","paper":"/paper/cdchat-a-large-multimodal-model-for-remote","title":"CDChat: A Large Multimodal Model for Remote Sensing Change Description","date":"2024-09-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"techmn/cdchat","path":"cdchat/eval/batch_cdchat_vqa.py","file_url":"https://github.com/techmn/cdchat/blob/HEAD/cdchat/eval/batch_cdchat_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2409.15657","paper":"/paper/mmpt-multimodal-prompt-tuning-for-zero-shot","title":"M$^2$PT: Multimodal Prompt Tuning for Zero-shot Instruction Learning","date":"2024-09-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"william-wang618/mmpt-emnlp2024","path":"M2PT/eval/model_vqa.py","file_url":"https://github.com/william-wang618/mmpt-emnlp2024/blob/HEAD/M2PT/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2409.14750","paper":"/paper/finecops-ref-a-new-dataset-and-task-for-fine","title":"FineCops-Ref: A new Dataset and Task for Fine-Grained Compositional Referring Expression Comprehension","date":"2024-09-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2409.14083","paper":"/paper/surf-teaching-large-vision-language-models-to","title":"SURf: Teaching Large Vision-Language Models to Selectively Utilize Retrieved Information","date":"2024-09-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"GasolSun36/SURf","path":"initial/generate_initial_data.py","file_url":"https://github.com/GasolSun36/SURf/blob/HEAD/initial/generate_initial_data.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2409.14083","paper":"/paper/surf-teaching-large-vision-language-models-to","title":"SURf: Teaching Large Vision-Language Models to Selectively Utilize Retrieved Information","date":"2024-09-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"GasolSun36/SURf","path":"eval/pope.py","file_url":"https://github.com/GasolSun36/SURf/blob/HEAD/eval/pope.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"06d40f4711e62357","mcp_get_code":{"code_sha256":"06d40f4711e62357"}},{"arxiv_id":"2409.14083","paper":"/paper/surf-teaching-large-vision-language-models-to","title":"SURf: Teaching Large Vision-Language Models to Selectively Utilize Retrieved Information","date":"2024-09-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"gasolsun36/surf","path":"initial/tool_evaluate.py","file_url":"https://github.com/gasolsun36/surf/blob/HEAD/initial/tool_evaluate.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"8a203abf91abfcad","mcp_get_code":{"code_sha256":"8a203abf91abfcad"}},{"arxiv_id":"2409.12953","paper":"/paper/journeybench-a-challenging-one-stop-vision","title":"JourneyBench: A Challenging One-Stop Vision-Language Understanding Benchmark of Generated Images","date":"2024-09-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"journeybench/journeybench","path":"automatic-qa-generator/baseline/llava_model_vcr.py","file_url":"https://github.com/journeybench/journeybench/blob/HEAD/automatic-qa-generator/baseline/llava_model_vcr.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2409.10994","paper":"/paper/less-is-more-a-simple-yet-effective-token","title":"Less is More: A Simple yet Effective Token Reduction Method for Efficient Multi-modal LLMs","date":"2024-09-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"freedomintelligence/trim","path":"llava/eval/model_vqa.py","file_url":"https://github.com/freedomintelligence/trim/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2409.10683","paper":"/paper/motif-motion-instruction-fine-tuning","title":"MotIF: Motion Instruction Fine-tuning","date":"2024-09-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Minyoung1005/motif","path":"LLaVA/llava/eval/model_vqa.py","file_url":"https://github.com/Minyoung1005/motif/blob/HEAD/LLaVA/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2409.10197","paper":"/paper/fit-and-prune-fast-and-training-free-visual","title":"Fit and Prune: Fast and Training-free Visual Token Pruning for Multi-modal Large Language Models","date":"2024-09-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ywh187/fitprune","path":"LLaVA_1.5/llava/eval/model_vqa.py","file_url":"https://github.com/ywh187/fitprune/blob/HEAD/LLaVA_1.5/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2409.06851","paper":"/paper/lime-m-less-is-more-for-evaluation-of-mllms","title":"LIME: Less Is More for MLLM Evaluation","date":"2024-09-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kangreen0210/lime","path":"llava_next_110B.py","file_url":"https://github.com/kangreen0210/lime/blob/HEAD/llava_next_110B.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2409.02919","paper":"/paper/hiprompt-tuning-free-higher-resolution","title":"HiPrompt: Tuning-free Higher-Resolution Generation with Hierarchical MLLM Prompts","date":"2024-09-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Liuxinyv/HiPrompt","path":"LLaVA/llava/eval/model_vqa.py","file_url":"https://github.com/Liuxinyv/HiPrompt/blob/HEAD/LLaVA/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2409.02889","paper":"/paper/longllava-scaling-multi-modal-llms-to-1000","title":"LongLLaVA: Scaling Multi-modal LLMs to 1000 Images Efficiently via a Hybrid Architecture","date":"2024-09-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"freedomintelligence/longllava","path":"benchmarks/MVBench/model_mvbench_qa.py","file_url":"https://github.com/freedomintelligence/longllava/blob/HEAD/benchmarks/MVBench/model_mvbench_qa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2409.01179","paper":"/paper/recoverable-compression-a-multimodal-vision","title":"Recoverable Compression: A Multimodal Vision Token Recovery Mechanism Guided by Text Information","date":"2024-09-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"banjiuyufen/Recoverable-Compression","path":"llava/eval/model_vqa.py","file_url":"https://github.com/banjiuyufen/Recoverable-Compression/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2409.00147","paper":"/paper/multimath-bridging-visual-and-mathematical","title":"MultiMath: Bridging Visual and Mathematical Reasoning for Large Language Models","date":"2024-08-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pengshuai-rin/multimath","path":"eval_mathverse/infer.py","file_url":"https://github.com/pengshuai-rin/multimath/blob/HEAD/eval_mathverse/infer.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2408.12902","paper":"/paper/iaa-inner-adaptor-architecture-empowers","title":"IAA: Inner-Adaptor Architecture Empowers Frozen Large Language Model with Multimodal Capabilities","date":"2024-08-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"360cvgroup/inner-adaptor-architecture","path":"iaa/eval/model_vqa_loader_llama3.py","file_url":"https://github.com/360cvgroup/inner-adaptor-architecture/blob/HEAD/iaa/eval/model_vqa_loader_llama3.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2408.08862","paper":"/paper/visual-agents-as-fast-and-slow-thinkers","title":"Visual Agents as Fast and Slow Thinkers","date":"2024-08-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"guangyans/sys2-llava","path":"ROILLaVA/llava/eval/model_vqa.py","file_url":"https://github.com/guangyans/sys2-llava/blob/HEAD/ROILLaVA/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2408.03735","paper":"/paper/advancing-multimodal-large-language-models","title":"Advancing Multimodal Large Language Models with Quantization-Aware Scale Learning for Efficient Adaptation","date":"2024-08-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xjjxmu/qslaw","path":"llava/eval/model_vqa.py","file_url":"https://github.com/xjjxmu/qslaw/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2408.02900","paper":"/paper/2408-02900","title":"MedTrinity-25M: A Large-scale Multimodal Dataset with Multigranular Annotations for Medicine","date":"2024-08-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"UCSC-VLAA/MedTrinity-25M","path":"llava/eval/model_vqa.py","file_url":"https://github.com/UCSC-VLAA/MedTrinity-25M/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2407.21771","paper":"/paper/paying-more-attention-to-image-a-training","title":"Paying More Attention to Image: A Training-Free Method for Alleviating Hallucination in LVLMs","date":"2024-07-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hasanar1f/llava-hallunication-fix","path":"modPAI/llava/eval/model_vqa.py","file_url":"https://github.com/hasanar1f/llava-hallunication-fix/blob/HEAD/modPAI/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2407.21439","paper":"/paper/2407-21439","title":"MLLM Is a Strong Reranker: Advancing Multimodal Retrieval-augmented Generation via Knowledge-enhanced Reranking and Noise-injected Training","date":"2024-07-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"idea-finai/ragllava","path":"llava/eval/model_vqa.py","file_url":"https://github.com/idea-finai/ragllava/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2407.06189","paper":"/paper/video-star-self-training-enables-video","title":"Video-STaR: Self-Training Enables Video Instruction Tuning with Any Supervision","date":"2024-07-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"orrzohar/Video-STaR","path":"videollava/eval/video/run_inference_video_qa.py","file_url":"https://github.com/orrzohar/Video-STaR/blob/HEAD/videollava/eval/video/run_inference_video_qa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2407.05131","paper":"/paper/rule-reliable-multimodal-rag-for-factuality","title":"RULE: Reliable Multimodal RAG for Factuality in Medical Vision Language Models","date":"2024-07-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"richard-peng-xia/rule","path":"povid_infer.py","file_url":"https://github.com/richard-peng-xia/rule/blob/HEAD/povid_infer.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2406.20098","paper":"/paper/web2code-a-large-scale-webpage-to-code","title":"Web2Code: A Large-scale Webpage-to-Code Dataset and Evaluation Framework for Multimodal LLMs","date":"2024-06-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"MBZUAI-LLM/web2code","path":"web2code/llava/eval/model_vqa.py","file_url":"https://github.com/MBZUAI-LLM/web2code/blob/HEAD/web2code/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2406.20095","paper":"/paper/llara-supercharging-robot-learning-data-for","title":"LLaRA: Supercharging Robot Learning Data for Vision-Language Policy","date":"2024-06-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"LostXine/LLaRA","path":"train-llava/llava/eval/model_vqa.py","file_url":"https://github.com/LostXine/LLaRA/blob/HEAD/train-llava/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2406.20092","paper":"/paper/llavolta-efficient-multi-modal-models-via","title":"Efficient Large Multi-modal Models via Visual Context Compression","date":"2024-06-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Beckschen/LLaVolta","path":"llava/eval/model_vqa.py","file_url":"https://github.com/Beckschen/LLaVolta/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2406.19973","paper":"/paper/stllava-med-self-training-large-language-and","title":"STLLaVA-Med: Self-Training Large Language and Vision Assistant for Medical Question-Answering","date":"2024-06-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"heliossun/stllava-med","path":"llava/eval/model_cap.py","file_url":"https://github.com/heliossun/stllava-med/blob/HEAD/llava/eval/model_cap.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2406.18139","paper":"/paper/look-m-look-once-optimization-in-kv-cache-for","title":"LOOK-M: Look-Once Optimization in KV Cache for Efficient Multimodal Long-Context Inference","date":"2024-06-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sustechbruce/look-m","path":"LLaVA-mix_merge_v1/llava/eval/model_vqa.py","file_url":"https://github.com/sustechbruce/look-m/blob/HEAD/LLaVA-mix_merge_v1/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2406.16860","paper":"/paper/cambrian-1-a-fully-open-vision-centric","title":"Cambrian-1: A Fully Open, Vision-Centric Exploration of Multimodal LLMs","date":"2024-06-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cambrian-mllm/cambrian","path":"eval/eval/omni/omni_eval.py","file_url":"https://github.com/cambrian-mllm/cambrian/blob/HEAD/eval/eval/omni/omni_eval.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0970b04a87d6849f","mcp_get_code":{"code_sha256":"0970b04a87d6849f"}},{"arxiv_id":"2406.16449","paper":"/paper/evaluating-and-analyzing-relationship","title":"Evaluating and Analyzing Relationship Hallucinations in Large Vision-Language Models","date":"2024-06-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mrwu-mac/R-Bench","path":"demos/eval_rbench.py","file_url":"https://github.com/mrwu-mac/R-Bench/blob/HEAD/demos/eval_rbench.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2406.14852","paper":"/paper/is-a-picture-worth-a-thousand-words-delving","title":"Is A Picture Worth A Thousand Words? Delving Into Spatial Reasoning for Vision Language Models","date":"2024-06-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"BAAI-DCAI/Bunny","path":"bunny/eval/model_vqa.py","file_url":"https://github.com/BAAI-DCAI/Bunny/blob/HEAD/bunny/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2406.14498","paper":"/paper/llasa-large-multimodal-agent-for-human","title":"LLaSA: A Multimodal LLM for Human Activity Analysis Through Wearable and Smartphone Sensors","date":"2024-06-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bashlab/llasa","path":"LLaSA/llava/eval/model_vqa.py","file_url":"https://github.com/bashlab/llasa/blob/HEAD/LLaSA/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2406.13642","paper":"/paper/spatialbot-precise-spatial-understanding-with","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","date":"2024-06-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"baai-dcai/spatialbot","path":"bunny/eval/model_vqa.py","file_url":"https://github.com/baai-dcai/spatialbot/blob/HEAD/bunny/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2406.13173","paper":"/paper/biomedical-visual-instruction-tuning-with","title":"Biomedical Visual Instruction Tuning with Clinician Preference Alignment","date":"2024-06-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mao1207/BioMed-VITAL","path":"eval/model_vqa_med.py","file_url":"https://github.com/mao1207/BioMed-VITAL/blob/HEAD/eval/model_vqa_med.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2406.12718","paper":"/paper/agla-mitigating-object-hallucinations-in","title":"AGLA: Mitigating Object Hallucinations in Large Vision-Language Models with Assembly of Global and Local Attention","date":"2024-06-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lackel/agla","path":"eval/run_instructblip.py","file_url":"https://github.com/lackel/agla/blob/HEAD/eval/run_instructblip.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2406.12275","paper":"/paper/voco-llama-towards-vision-compression-with","title":"VoCo-LLaMA: Towards Vision Compression with Large Language Models","date":"2024-06-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Yxxxb/VoCo-LLaMA","path":"llava/eval/model_vqa_loader.py","file_url":"https://github.com/Yxxxb/VoCo-LLaMA/blob/HEAD/llava/eval/model_vqa_loader.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2406.11823","paper":"/paper/on-efficient-language-and-vision-assistants","title":"On Efficient Language and Vision Assistants for Visually-Situated Natural Language Understanding: What Matters in Reading and Reasoning","date":"2024-06-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"naver-ai/elva","path":"Elva/model_vqa_parsing.py","file_url":"https://github.com/naver-ai/elva/blob/HEAD/Elva/model_vqa_parsing.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"8a203abf91abfcad","mcp_get_code":{"code_sha256":"8a203abf91abfcad"}},{"arxiv_id":"2406.11280","paper":"/paper/i-srt-aligning-large-multimodal-models-for","title":"ISR-DPO: Aligning Large Multimodal Models for Videos by Iterative Self-Retrospective DPO","date":"2024-06-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"snumprlab/SRT","path":"llava/eval/model_vqa.py","file_url":"https://github.com/snumprlab/SRT/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2406.10638","paper":"/paper/seeing-clearly-answering-incorrectly-a","title":"Unveiling the Ignorance of MLLMs: Seeing Clearly, Answering Incorrectly","date":"2024-06-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"baai-dcai/multimodal-robustness-benchmark","path":"evaluation/evaluation/model_vqa_mmr.py","file_url":"https://github.com/baai-dcai/multimodal-robustness-benchmark/blob/HEAD/evaluation/evaluation/model_vqa_mmr.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2406.10100","paper":"/paper/skysensegpt-a-fine-grained-instruction-tuning","title":"SkySenseGPT: A Fine-Grained Instruction Tuning Dataset and Model for Remote Sensing Vision-Language Understanding","date":"2024-06-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"luo-z13/skysensegpt","path":"Eval_scripts/batch_fitrsrc_single_choice_qa.py","file_url":"https://github.com/luo-z13/skysensegpt/blob/HEAD/Eval_scripts/batch_fitrsrc_single_choice_qa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2406.09400","paper":"/paper/yo-llava-your-personalized-language-and","title":"Yo'LLaVA: Your Personalized Language and Vision Assistant","date":"2024-06-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"WisconsinAIVision/YoLLaVA","path":"llava/eval/model_vqa.py","file_url":"https://github.com/WisconsinAIVision/YoLLaVA/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2406.09367","paper":"/paper/needle-in-a-video-haystack-a-scalable","title":"Needle In A Video Haystack: A Scalable Synthetic Evaluator for Video MLLMs","date":"2024-06-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"joez17/videoniah","path":"llava_examples/model_video_niah.py","file_url":"https://github.com/joez17/videoniah/blob/HEAD/llava_examples/model_video_niah.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2406.08487","paper":"/paper/beyond-llava-hd-diving-into-high-resolution","title":"Beyond LLaVA-HD: Diving into High-Resolution Large Multimodal Models","date":"2024-06-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yfzhang114/slime","path":"llava/eval/model_vqa.py","file_url":"https://github.com/yfzhang114/slime/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2406.08100","paper":"/paper/multimodal-table-understanding","title":"Multimodal Table Understanding","date":"2024-06-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"spursgozmy/table-llava","path":"llava/eval/model_vqa.py","file_url":"https://github.com/spursgozmy/table-llava/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2406.02884","paper":"/paper/posterllava-constructing-a-unified-multi","title":"PosterLLaVa: Constructing a Unified Multi-modal Layout Generator with LLM","date":"2024-06-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"posterllava/posterllava","path":"llava/eval/model_vqa.py","file_url":"https://github.com/posterllava/posterllava/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2405.19315","paper":"/paper/matryoshka-query-transformer-for-large-vision","title":"Matryoshka Query Transformer for Large Vision-Language Models","date":"2024-05-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"gordonhu608/mqt-llava","path":"llava/eval/model_vqa.py","file_url":"https://github.com/gordonhu608/mqt-llava/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2405.18654","paper":"/paper/mitigating-object-hallucination-via-data","title":"Data-augmented phrase-level alignment for mitigating object hallucination","date":"2024-05-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pritamqu/HALVA","path":"eval_hall/model_amber_loader.py","file_url":"https://github.com/pritamqu/HALVA/blob/HEAD/eval_hall/model_amber_loader.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2405.18415","paper":"/paper/why-are-visually-grounded-language-models-bad","title":"Why are Visually-Grounded Language Models Bad at Image Classification?","date":"2024-05-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yuhui-zh15/vlmclassifier","path":"training_analysis/llava/model_vqa.py","file_url":"https://github.com/yuhui-zh15/vlmclassifier/blob/HEAD/training_analysis/llava/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2405.18406","paper":"/paper/raccoon-remove-add-and-change-video-content","title":"RACCooN: A Versatile Instructional Video Editing Framework with Auto-Generated Narratives","date":"2024-05-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jaehong31/raccoon","path":"p2v/LLaVA/llava/eval/model_vqa.py","file_url":"https://github.com/jaehong31/raccoon/blob/HEAD/p2v/LLaVA/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2405.15973","paper":"/paper/enhancing-visual-language-modality-alignment","title":"Enhancing Visual-Language Modality Alignment in Large Vision Language Models via Self-Improvement","date":"2024-05-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"umd-huang-lab/sima","path":"llava/sima/model_self_rewarding.py","file_url":"https://github.com/umd-huang-lab/sima/blob/HEAD/llava/sima/model_self_rewarding.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2405.14974","paper":"/paper/lova3-learning-to-visual-question-answering","title":"LOVA3: Learning to Visual Question Answering, Asking and Assessment","date":"2024-05-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"showlab/LOVA3","path":"llava/eval/model_vqa.py","file_url":"https://github.com/showlab/LOVA3/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2405.12107","paper":"/paper/imp-highly-capable-large-multimodal-models","title":"Imp: Highly Capable Large Multimodal Models for Mobile Devices","date":"2024-05-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"milvlg/imp","path":"imp_llava/eval/model_vqa.py","file_url":"https://github.com/milvlg/imp/blob/HEAD/imp_llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2405.07798","paper":"/paper/freeva-offline-mllm-as-training-free-video","title":"FreeVA: Offline MLLM as Training-Free Video Assistant","date":"2024-05-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"whwu95/freeva","path":"llava/eval/run_inference_benchmark_consistency.py","file_url":"https://github.com/whwu95/freeva/blob/HEAD/llava/eval/run_inference_benchmark_consistency.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2405.05949","paper":"/paper/cumo-scaling-multimodal-llm-with-co-upcycled","title":"CuMo: Scaling Multimodal LLM with Co-Upcycled Mixture-of-Experts","date":"2024-05-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shi-labs/cumo","path":"cumo/eval/model_vqa.py","file_url":"https://github.com/shi-labs/cumo/blob/HEAD/cumo/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2404.13046","paper":"/paper/mova-adapting-mixture-of-vision-experts-to","title":"MoVA: Adapting Mixture of Vision Experts to Multimodal Context","date":"2024-04-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"templex98/mova","path":"mova/eval/model_vqa.py","file_url":"https://github.com/templex98/mova/blob/HEAD/mova/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2404.10501","paper":"/paper/self-supervised-visual-preference-alignment","title":"Self-Supervised Visual Preference Alignment","date":"2024-04-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Kevinz-code/SeVa","path":"seva/llava/eval/model_vqa.py","file_url":"https://github.com/Kevinz-code/SeVa/blob/HEAD/seva/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"GPL-3.0","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2404.10237","paper":"/paper/moe-tinymed-mixture-of-experts-for-tiny","title":"Med-MoE: Mixture of Domain-Specific Experts for Lightweight Medical Vision-Language Models","date":"2024-04-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jiangsongtao/tinymed","path":"model_vqa_med.py","file_url":"https://github.com/jiangsongtao/tinymed/blob/HEAD/model_vqa_med.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2404.09027","paper":"/paper/ming-moe-enhancing-medical-multi-task","title":"MING-MOE: Enhancing Medical Multi-Task Learning in Large Language Models with Sparse Mixture of Low-Rank Adapter Experts","date":"2024-04-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mediabrain-sjtu/medicalgpt-zh","path":"ming/eval/model_diverse_gen.py","file_url":"https://github.com/mediabrain-sjtu/medicalgpt-zh/blob/HEAD/ming/eval/model_diverse_gen.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2404.05892","paper":"/paper/eagle-and-finch-rwkv-with-matrix-valued","title":"Eagle and Finch: RWKV with Matrix-Valued States and Dynamic Recurrence","date":"2024-04-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"howard-hou/visualrwkv","path":"VisualRWKV-v7/v7.10/evaluate_imagenet.py","file_url":"https://github.com/howard-hou/visualrwkv/blob/HEAD/VisualRWKV-v7/v7.10/evaluate_imagenet.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2403.18814","paper":"/paper/mini-gemini-mining-the-potential-of-multi","title":"Mini-Gemini: Mining the Potential of Multi-modality Vision Language Models","date":"2024-03-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dvlab-research/minigemini","path":"mgm/eval/model_math_vista.py","file_url":"https://github.com/dvlab-research/minigemini/blob/HEAD/mgm/eval/model_math_vista.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2403.18775","paper":"/paper/imagenet-d-benchmarking-neural-network","title":"ImageNet-D: Benchmarking Neural Network Robustness on Diffusion Synthetic Object","date":"2024-03-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"chenshuang-zhang/imagenet_d","path":"LLaVA/llava/eval/model_vqa.py","file_url":"https://github.com/chenshuang-zhang/imagenet_d/blob/HEAD/LLaVA/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2403.18252","paper":"/paper/beyond-embeddings-the-promise-of-visual-table","title":"Beyond Embeddings: The Promise of Visual Table in Visual Reasoning","date":"2024-03-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lavi-lab/visual-table","path":"llava/eval/model_vqa.py","file_url":"https://github.com/lavi-lab/visual-table/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2403.16999","paper":"/paper/visual-cot-unleashing-chain-of-thought","title":"Visual CoT: Advancing Multi-Modal Language Models with a Comprehensive Dataset and Benchmark for Chain-of-Thought Reasoning","date":"2024-03-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"deepcs233/visual-cot","path":"llava/eval/model_cot_det_loader.py","file_url":"https://github.com/deepcs233/visual-cot/blob/HEAD/llava/eval/model_cot_det_loader.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2403.16998","paper":"/paper/understanding-long-videos-in-one-multimodal","title":"Understanding Long Videos with Multimodal Language Models","date":"2024-03-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kahnchana/mvu","path":"src/model_frame_selection.py","file_url":"https://github.com/kahnchana/mvu/blob/HEAD/src/model_frame_selection.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2403.12966","paper":"/paper/chain-of-spot-interactive-reasoning-improves","title":"Chain-of-Spot: Interactive Reasoning Improves Large Vision-Language Models","date":"2024-03-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dongyh20/chain-of-spot","path":"llava/eval/model_vqa.py","file_url":"https://github.com/dongyh20/chain-of-spot/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2403.08002","paper":"/paper/training-small-multimodal-models-to-bridge","title":"Towards a clinically accessible radiology foundation model: open-access and lightweight, with automated evaluation","date":"2024-03-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"microsoft/llava-rad","path":"llava/eval/model_mimic_cxr.py","file_url":"https://github.com/microsoft/llava-rad/blob/HEAD/llava/eval/model_mimic_cxr.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2403.05262","paper":"/paper/debiasing-large-visual-language-models","title":"Debiasing Multimodal Large Language Models","date":"2024-03-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2403.03003","paper":"/paper/feast-your-eyes-mixture-of-resolution","title":"Feast Your Eyes: Mixture-of-Resolution Adaptation for Multimodal Large Language Models","date":"2024-03-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"luogen1996/llava-hr","path":"llava_hr/eval/model_caption_loader.py","file_url":"https://github.com/luogen1996/llava-hr/blob/HEAD/llava_hr/eval/model_caption_loader.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2402.19474","paper":"/paper/the-all-seeing-project-v2-towards-general","title":"The All-Seeing Project V2: Towards General Relation Comprehension of the Open World","date":"2024-02-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"opengvlab/all-seeing","path":"all-seeing-v2/llava/eval/model_vqa.py","file_url":"https://github.com/opengvlab/all-seeing/blob/HEAD/all-seeing-v2/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2402.16050","paper":"/paper/lstp-language-guided-spatial-temporal-prompt","title":"Efficient Temporal Extrapolation of Multimodal Large Language Models with Temporal Grounding Bridge","date":"2024-02-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bigai-nlco/VideoTGB","path":"eval/inference.py","file_url":"https://github.com/bigai-nlco/VideoTGB/blob/HEAD/eval/inference.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2402.14545","paper":"/paper/less-is-more-mitigating-multimodal","title":"Less is More: Mitigating Multimodal Hallucination from an EOS Decision Perspective","date":"2024-02-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yuezih/less-is-more","path":"LLaVA/llava/eval/model_vqa.py","file_url":"https://github.com/yuezih/less-is-more/blob/HEAD/LLaVA/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2402.14289","paper":"/paper/tinyllava-a-framework-of-small-scale-large","title":"TinyLLaVA: A Framework of Small-scale Large Multimodal Models","date":"2024-02-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dlcv-buaa/tinyllavabench","path":"tinyllava/eval/model_vqa.py","file_url":"https://github.com/dlcv-buaa/tinyllavabench/blob/HEAD/tinyllava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2402.11941","paper":"/paper/comprehensive-cognitive-llm-agent-for","title":"CoCo-Agent: A Comprehensive Cognitive MLLM Agent for Smartphone GUI Automation","date":"2024-02-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xbmxb/coco-agent","path":"llava/eval/model_aitw.py","file_url":"https://github.com/xbmxb/coco-agent/blob/HEAD/llava/eval/model_aitw.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2402.11530","paper":"/paper/efficient-multimodal-learning-from-data","title":"Efficient Multimodal Learning from Data-centric Perspective","date":"2024-02-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"baai-dcai/bunny","path":"bunny/eval/model_vqa.py","file_url":"https://github.com/baai-dcai/bunny/blob/HEAD/bunny/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2402.11411","paper":"/paper/aligning-modalities-in-vision-large-language","title":"Aligning Modalities in Vision Large Language Models via Preference Fine-tuning","date":"2024-02-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yiyangzhou/povid","path":"povid_infer.py","file_url":"https://github.com/yiyangzhou/povid/blob/HEAD/povid_infer.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2402.10884","paper":"/paper/multi-modal-preference-alignment-remedies","title":"Multi-modal Preference Alignment Remedies Degradation of Visual Instruction Tuning on Language Models","date":"2024-02-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"findalexli/mllm-dpo","path":"llava/eval/model_vqa.py","file_url":"https://github.com/findalexli/mllm-dpo/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2402.03766","paper":"/paper/2402-03766","title":"MobileVLM V2: Faster and Stronger Baseline for Vision Language Model","date":"2024-02-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"meituan-automl/mobilevlm","path":"mobilevlm/eval/model_vqa_loader.py","file_url":"https://github.com/meituan-automl/mobilevlm/blob/HEAD/mobilevlm/eval/model_vqa_loader.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2402.03190","paper":"/paper/unified-hallucination-detection-for","title":"Unified Hallucination Detection for Multimodal Large Language Models","date":"2024-02-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zjunlp/easydetect","path":"HalDet-LLaVA/llava/eval/model_vqa.py","file_url":"https://github.com/zjunlp/easydetect/blob/HEAD/HalDet-LLaVA/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2401.06591","paper":"/paper/prometheus-vision-vision-language-model-as-a","title":"Prometheus-Vision: Vision-Language Model as a Judge for Fine-Grained Evaluation","date":"2024-01-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kaistai/prometheus-vision","path":"llava/eval/model_vqa.py","file_url":"https://github.com/kaistai/prometheus-vision/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2401.06209","paper":"/paper/eyes-wide-shut-exploring-the-visual","title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","date":"2024-01-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tsb0601/MMVP","path":"LLaVA/llava/eval/model_vqa.py","file_url":"https://github.com/tsb0601/MMVP/blob/HEAD/LLaVA/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2401.05827","paper":"/paper/hallucination-benchmark-in-medical-visual","title":"Hallucination Benchmark in Medical Visual Question Answering","date":"2024-01-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"knowlab/halt-medvqa","path":"model_vqa_halt.py","file_url":"https://github.com/knowlab/halt-medvqa/blob/HEAD/model_vqa_halt.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2401.02330","paper":"/paper/llava-ph-efficient-multi-modal-assistant-with","title":"LLaVA-Phi: Efficient Multi-Modal Assistant with Small Language Model","date":"2024-01-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zhuyiche/llava-phi","path":"llava_phi/eval/model_vqa.py","file_url":"https://github.com/zhuyiche/llava-phi/blob/HEAD/llava_phi/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2312.17243","paper":"/paper/unsupervised-universal-image-segmentation","title":"Unsupervised Universal Image Segmentation","date":"2023-12-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dantong88/llarva","path":"llava/eval/model_vqa.py","file_url":"https://github.com/dantong88/llarva/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2312.14233","paper":"/paper/vcoder-versatile-vision-encoders-for","title":"VCoder: Versatile Vision Encoders for Multimodal Large Language Models","date":"2023-12-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shi-labs/vcoder","path":"vcoder_llava/eval/model_depth_loader.py","file_url":"https://github.com/shi-labs/vcoder/blob/HEAD/vcoder_llava/eval/model_depth_loader.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2312.11370","paper":"/paper/g-llava-solving-geometric-problem-with-multi","title":"G-LLaVA: Solving Geometric Problem with Multi-Modal Large Language Model","date":"2023-12-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pipilurj/g-llava","path":"gllava/eval/model_vqa_llava.py","file_url":"https://github.com/pipilurj/g-llava/blob/HEAD/gllava/eval/model_vqa_llava.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2312.06731","paper":"/paper/genixer-empowering-multimodal-large-language","title":"Genixer: Empowering Multimodal Large Language Models as a Powerful Data Generator","date":"2023-12-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zhaohengyuan1/genixer","path":"Genixer_LLaVA/model_genixer_eval.py","file_url":"https://github.com/zhaohengyuan1/genixer/blob/HEAD/Genixer_LLaVA/model_genixer_eval.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2312.04746","paper":"/paper/quilt-llava-visual-instruction-tuning-by","title":"Quilt-LLaVA: Visual Instruction Tuning by Extracting Localized Narratives from Open-Source Histopathology Videos","date":"2023-12-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"aldraus/quilt-llava","path":"llava/eval/model_vqa.py","file_url":"https://github.com/aldraus/quilt-llava/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2312.02949","paper":"/paper/llava-grounding-grounded-visual-chat-with","title":"LLaVA-Grounding: Grounded Visual Chat with Large Multimodal Models","date":"2023-12-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ux-decoder/llava-grounding","path":"llava/eval/model_vqa.py","file_url":"https://github.com/ux-decoder/llava-grounding/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2312.02896","paper":"/paper/benchlmm-benchmarking-cross-style-visual","title":"BenchLMM: Benchmarking Cross-style Visual Capability of Large Multimodal Models","date":"2023-12-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"aifeg/benchgpt","path":"baseline/LLaVA/BenchGPT_LLaVA_model_vqa.py","file_url":"https://github.com/aifeg/benchgpt/blob/HEAD/baseline/LLaVA/BenchGPT_LLaVA_model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2311.17076","paper":"/paper/compositional-chain-of-thought-prompting-for","title":"Compositional Chain-of-Thought Prompting for Large Multimodal Models","date":"2023-11-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"chancharikmitra/ccot","path":"GPT-4V/Sphinx_bench.py","file_url":"https://github.com/chancharikmitra/ccot/blob/HEAD/GPT-4V/Sphinx_bench.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2311.17043","paper":"/paper/llama-vid-an-image-is-worth-2-tokens-in-large","title":"LLaMA-VID: An Image is Worth 2 Tokens in Large Language Models","date":"2023-11-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dvlab-research/llama-vid","path":"llamavid/eval/model_activitynet_qa.py","file_url":"https://github.com/dvlab-research/llama-vid/blob/HEAD/llamavid/eval/model_activitynet_qa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2311.16922","paper":"/paper/mitigating-object-hallucinations-in-large","title":"Mitigating Object Hallucinations in Large Vision-Language Models through Visual Contrastive Decoding","date":"2023-11-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"damo-nlp-sg/vcd","path":"experiments/eval/object_hallucination_vqa_instructblip.py","file_url":"https://github.com/damo-nlp-sg/vcd/blob/HEAD/experiments/eval/object_hallucination_vqa_instructblip.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2311.16502","paper":"/paper/mmmu-a-massive-multi-discipline-multimodal","title":"MMMU: A Massive Multi-discipline Multimodal Understanding and Reasoning Benchmark for Expert AGI","date":"2023-11-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"eric-ai-lab/probmed","path":"eval/inference/CheXagent/model_vqa_med.py","file_url":"https://github.com/eric-ai-lab/probmed/blob/HEAD/eval/inference/CheXagent/model_vqa_med.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2311.07362","paper":"/paper/volcano-mitigating-multimodal-hallucination","title":"Volcano: Mitigating Multimodal Hallucination through Self-Feedback Guided Revision","date":"2023-11-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kaistai/volcano","path":"llava/eval/volcano_pope.py","file_url":"https://github.com/kaistai/volcano/blob/HEAD/llava/eval/volcano_pope.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2311.01477","paper":"/paper/faithscore-evaluating-hallucinations-in-large","title":"FaithScore: Fine-grained Evaluations of Hallucinations in Large Vision-Language Models","date":"2023-11-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bcdnlp/faithscore","path":"src/faithscore/llava15.py","file_url":"https://github.com/bcdnlp/faithscore/blob/HEAD/src/faithscore/llava15.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2310.08588","paper":"/paper/octopus-embodied-vision-language-programmer","title":"Octopus: Embodied Vision-Language Programmer from Environmental Feedback","date":"2023-10-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dongyh20/octopus","path":"octopus/LLaVA/llava/eval/model_vqa.py","file_url":"https://github.com/dongyh20/octopus/blob/HEAD/octopus/LLaVA/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2310.00582","paper":"/paper/pink-unveiling-the-power-of-referential","title":"Pink: Unveiling the Power of Referential Comprehension for Multi-modal LLMs","date":"2023-10-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sy-xuan/pink","path":"pink/eval/model_gqa.py","file_url":"https://github.com/sy-xuan/pink/blob/HEAD/pink/eval/model_gqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2308.13566","paper":"/paper/mllm-dataengine-an-iterative-refinement","title":"MLLM-DataEngine: An Iterative Refinement Approach for MLLM","date":"2023-08-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"opendatalab/mllm-dataengine","path":"LLaVA/llava/eval/model_vqa.py","file_url":"https://github.com/opendatalab/mllm-dataengine/blob/HEAD/LLaVA/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2308.10253","paper":"/paper/stablellava-enhanced-visual-instruction","title":"StableLLaVA: Enhanced Visual Instruction Tuning with Synthesized Image-Dialogue Data","date":"2023-08-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"icoz69/stablellava","path":"llava/model_vqa_mm.py","file_url":"https://github.com/icoz69/stablellava/blob/HEAD/llava/model_vqa_mm.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2308.03349","paper":"/paper/scigraphqa-a-large-scale-synthetic-multi-turn","title":"SciGraphQA: A Large-Scale Synthetic Multi-Turn Question-Answering Dataset for Scientific Graphs","date":"2023-08-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"findalexli/SciGraphQA","path":"llava/eval/model_vqa.py","file_url":"https://github.com/findalexli/SciGraphQA/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2307.10490","paper":"/paper/ab-using-images-and-sounds-for-indirect","title":"Abusing Images and Sounds for Indirect Instruction Injection in Multi-Modal LLMs","date":"2023-07-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ebagdasa/multimodal_injection","path":"llava/eval/model_vqa.py","file_url":"https://github.com/ebagdasa/multimodal_injection/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2306.17107","paper":"/paper/llavar-enhanced-visual-instruction-tuning-for","title":"LLaVAR: Enhanced Visual Instruction Tuning for Text-Rich Image Understanding","date":"2023-06-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2305.17852","paper":"/paper/hierarchical-neural-memory-network-for-low-1","title":"Hierarchical Neural Memory Network for Low Latency Event Processing","date":"2023-05-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hamarh/HMNet_pth","path":"hmnet/utils/common.py","file_url":"https://github.com/hamarh/HMNet_pth/blob/HEAD/hmnet/utils/common.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"0343b784110cc40e","mcp_get_code":{"code_sha256":"0343b784110cc40e"}},{"arxiv_id":"2112.05682","paper":"/paper/self-attention-does-not-need-o-n-2-memory","title":"Self-attention Does Not Need $O(n^2)$ Memory","date":"2021-12-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"X-iZhang/Libra","path":"libra/eval/eval_vqa_libra.py","file_url":"https://github.com/X-iZhang/Libra/blob/HEAD/libra/eval/eval_vqa_libra.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"openreview_XtIRCAEYoJ","paper":null,"title":"arXiv:openreview_XtIRCAEYoJ","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"WayneTomas/Artemis","path":"infer_artemis.py","file_url":"https://github.com/WayneTomas/Artemis/blob/HEAD/infer_artemis.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"9d3ee2935018d3b1","mcp_get_code":{"code_sha256":"9d3ee2935018d3b1"}},{"arxiv_id":"openreview_XtIRCAEYoJ","paper":null,"title":"arXiv:openreview_XtIRCAEYoJ","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"WayneTomas/Artemis","path":"val/coco_detection/infer_artemis_detection.py","file_url":"https://github.com/WayneTomas/Artemis/blob/HEAD/val/coco_detection/infer_artemis_detection.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"d5f982b3efa4af6a","mcp_get_code":{"code_sha256":"d5f982b3efa4af6a"}},{"arxiv_id":"Zhong_AIM_Adaptive_Inference_of_Multi-Modal_LLMs_via_Token_Merging_and_ICCV_2025_paper","paper":null,"title":"arXiv:Zhong_AIM_Adaptive_Inference_of_Multi-Modal_LLMs_via_Token_Merging_and_ICCV_2025_paper","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"LaVi-Lab/AIM","path":"llava/eval/model_vqa.py","file_url":"https://github.com/LaVi-Lab/AIM/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"Xing_Conical_Visual_Concentration_for_Efficient_Large_Vision-Language_Models_CVPR_2025_paper","paper":null,"title":"arXiv:Xing_Conical_Visual_Concentration_for_Efficient_Large_Vision-Language_Models_CVPR_2025_paper","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"Cooperx521/PyramidDrop","path":"llava/eval/model_vqa.py","file_url":"https://github.com/Cooperx521/PyramidDrop/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"Jin_Chat-UniVi_Unified_Visual_Representation_Empowers_Large_Language_Models_with_Image_CVPR_2024_paper","paper":null,"title":"arXiv:Jin_Chat-UniVi_Unified_Visual_Representation_Empowers_Large_Language_Models_with_Image_CVPR_2024_paper","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"PKU-YuanGroup/Chat-UniVi","path":"ChatUniVi/eval/model_coco_vqa.py","file_url":"https://github.com/PKU-YuanGroup/Chat-UniVi/blob/HEAD/ChatUniVi/eval/model_coco_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2025.naacl-long.579","paper":null,"title":"arXiv:2025.naacl-long.579","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"DAMO-NLP-SG/VideoLLaMA2","path":"videollama2/eval/inference_video_oqa_activitynet.py","file_url":"https://github.com/DAMO-NLP-SG/VideoLLaMA2/blob/HEAD/videollama2/eval/inference_video_oqa_activitynet.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2025.findings-emnlp.1095","paper":null,"title":"arXiv:2025.findings-emnlp.1095","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"AngelAlita/AsD","path":"llava/eval/model_vqa.py","file_url":"https://github.com/AngelAlita/AsD/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2025.findings-acl.865","paper":null,"title":"arXiv:2025.findings-acl.865","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"DeepLearnXMU/AVG-LLaVA","path":"llava/eval/model_vqa.py","file_url":"https://github.com/DeepLearnXMU/AVG-LLaVA/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2025.findings-acl.359","paper":null,"title":"arXiv:2025.findings-acl.359","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"G-JWLee/TAMP","path":"llava/eval/model_vqa.py","file_url":"https://github.com/G-JWLee/TAMP/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2025.findings-acl.327","paper":null,"title":"arXiv:2025.findings-acl.327","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"SakuraTroyChen/PyPE","path":"LLaVA/llava/eval/model_vqa.py","file_url":"https://github.com/SakuraTroyChen/PyPE/blob/HEAD/LLaVA/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2024.findings-naacl.226","paper":null,"title":"arXiv:2024.findings-naacl.226","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"nguyennm1024/OSCaR","path":"llava/eval/model_vqa.py","file_url":"https://github.com/nguyennm1024/OSCaR/blob/HEAD/llava/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2024.findings-emnlp.775","paper":null,"title":"arXiv:2024.findings-emnlp.775","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"YuxiXie/V-DPO","path":"llava_dpo/eval/model_vqa.py","file_url":"https://github.com/YuxiXie/V-DPO/blob/HEAD/llava_dpo/eval/model_vqa.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2024.findings-emnlp.619","paper":null,"title":"arXiv:2024.findings-emnlp.619","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"BlueZeros/MedCare","path":"medcare/eval/model_diverse_gen_batch.py","file_url":"https://github.com/BlueZeros/MedCare/blob/HEAD/medcare/eval/model_diverse_gen_batch.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2024.findings-emnlp.290","paper":null,"title":"arXiv:2024.findings-emnlp.290","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"bcdnlp/FAITHSCORE","path":"src/faithscore/llava15.py","file_url":"https://github.com/bcdnlp/FAITHSCORE/blob/HEAD/src/faithscore/llava15.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}},{"arxiv_id":"2024.findings-emnlp.221","paper":null,"title":"arXiv:2024.findings-emnlp.221","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"jiangsongtao/Med-MoE","path":"model_vqa_med.py","file_url":"https://github.com/jiangsongtao/Med-MoE/blob/HEAD/model_vqa_med.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"42a46570620cd9fa","mcp_get_code":{"code_sha256":"42a46570620cd9fa"}}]}