{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/unpad-image","entry":"unpad_image","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":72,"n_papers_ran":69,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":8,"n_samples_ran":5,"n_samples_fingerprinted":5,"n_places":73,"n_places_pointer_only":27,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":4,"ran":1,"unverified":3},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2606.00535","paper":"/paper/arxiv-2606-00535","title":"DREAM-S: Speculative Decoding with Searchable Drafting and Target-Aware Refinement for Multimodal Generation","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"SAI-Lab-NYU/DREAM-S","path":"dream_s/model/modeling_llava_next.py","file_url":"https://github.com/SAI-Lab-NYU/DREAM-S/blob/HEAD/dream_s/model/modeling_llava_next.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"31c6472fe907e4f3","mcp_get_code":{"code_sha256":"31c6472fe907e4f3"}},{"arxiv_id":"2604.13565","paper":"/paper/arxiv-2604-13565","title":"UHR-BAT: Budget-Aware Token Compression Vision-Language model for Ultra-High-Resolution Remote Sensing","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"Yunkaidang/UHR","path":"with-SAM/longva/longva/model/llava_arch.py","file_url":"https://github.com/Yunkaidang/UHR/blob/HEAD/with-SAM/longva/longva/model/llava_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"7606525af238fb64","mcp_get_code":{"code_sha256":"7606525af238fb64"}},{"arxiv_id":"2602.23699","paper":"/paper/arxiv-2602-23699","title":"HiDrop: Hierarchical Vision Token Reduction in MLLMs via Late Injection, Concave Pyramid Pruning, and Early Exit","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"EIT-NLP/HiDrop","path":"llava/model/llava_arch.py","file_url":"https://github.com/EIT-NLP/HiDrop/blob/HEAD/llava/model/llava_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"55c32993da87759b","mcp_get_code":{"code_sha256":"55c32993da87759b"}},{"arxiv_id":"2601.17868","paper":"/paper/arxiv-2601-17868","title":"VidLaDA: Bidirectional Diffusion Large Language Models for Efficient Video Understanding","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"ziHoHe/VidLaDA","path":"train/llava/model/llava_arch.py","file_url":"https://github.com/ziHoHe/VidLaDA/blob/HEAD/train/llava/model/llava_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"7606525af238fb64","mcp_get_code":{"code_sha256":"7606525af238fb64"}},{"arxiv_id":"2507.12455","paper":"/paper/mitigating-object-hallucinations-via-sentence","title":"Mitigating Object Hallucinations via Sentence-Level Early Intervention","date":"2025-07-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pspdada/SENTINEL","path":"llava/model/llava_arch.py","file_url":"https://github.com/pspdada/SENTINEL/blob/HEAD/llava/model/llava_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"55c32993da87759b","mcp_get_code":{"code_sha256":"55c32993da87759b"}},{"arxiv_id":"2507.00505","paper":"/paper/llava-sp-enhancing-visual-representation-with","title":"LLaVA-SP: Enhancing Visual Representation with Visual Spatial Tokens for MLLMs","date":"2025-07-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"CnFaker/LLaVA-SP","path":"llava/model/llava_arch.py","file_url":"https://github.com/CnFaker/LLaVA-SP/blob/HEAD/llava/model/llava_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"7606525af238fb64","mcp_get_code":{"code_sha256":"7606525af238fb64"}},{"arxiv_id":"2506.17202","paper":"/paper/unifork-exploring-modality-alignment-for","title":"UniFork: Exploring Modality Alignment for Unified Multimodal Understanding and Generation","date":"2025-06-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tliby/unifork","path":"unifork/model/llava_arch.py","file_url":"https://github.com/tliby/unifork/blob/HEAD/unifork/model/llava_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"55c32993da87759b","mcp_get_code":{"code_sha256":"55c32993da87759b"}},{"arxiv_id":"2506.10967","paper":"/paper/beyond-attention-or-similarity-maximizing","title":"Beyond Attention or Similarity: Maximizing Conditional Diversity for Token Pruning in MLLMs","date":"2025-06-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"theia-4869/cdpruner","path":"llava/model/llava_arch.py","file_url":"https://github.com/theia-4869/cdpruner/blob/HEAD/llava/model/llava_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"55c32993da87759b","mcp_get_code":{"code_sha256":"55c32993da87759b"}},{"arxiv_id":"2505.19235","paper":"/paper/corematching-a-co-adaptive-sparse-inference","title":"CoreMatching: A Co-adaptive Sparse Inference Framework with Token and Neuron Pruning for Comprehensive Acceleration of Vision-Language Models","date":"2025-05-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"55c32993da87759b","mcp_get_code":{"code_sha256":"55c32993da87759b"}},{"arxiv_id":"2505.14640","paper":"/paper/videoeval-pro-robust-and-realistic-long-video","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","date":"2025-05-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"opengvlab/videochat-flash","path":"llava-train_videochat/llava/model/llava_arch.py","file_url":"https://github.com/opengvlab/videochat-flash/blob/HEAD/llava-train_videochat/llava/model/llava_arch.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2c7b71210b8a363f","mcp_get_code":{"code_sha256":"2c7b71210b8a363f"}},{"arxiv_id":"2504.13169","paper":"/paper/generate-but-verify-reducing-hallucination-in","title":"Generate, but Verify: Reducing Hallucination in Vision-Language Models with Retrospective Resampling","date":"2025-04-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"55c32993da87759b","mcp_get_code":{"code_sha256":"55c32993da87759b"}},{"arxiv_id":"2504.02438","paper":"/paper/scaling-video-language-models-to-10k-frames","title":"Scaling Video-Language Models to 10K Frames via Hierarchical Differential Distillation","date":"2025-04-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"7606525af238fb64","mcp_get_code":{"code_sha256":"7606525af238fb64"}},{"arxiv_id":"2504.01934","paper":"/paper/illume-illuminating-unified-mllm-with-dual","title":"ILLUME+: Illuminating Unified MLLM with Dual Visual Tokenization and Diffusion Refinement","date":"2025-04-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"illume-unified-mllm/ILLUME_plus","path":"ILLUME/illume/model/illume_arch.py","file_url":"https://github.com/illume-unified-mllm/ILLUME_plus/blob/HEAD/ILLUME/illume/model/illume_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"55c32993da87759b","mcp_get_code":{"code_sha256":"55c32993da87759b"}},{"arxiv_id":"2504.01328","paper":"/paper/slow-fast-architecture-for-video-multi-modal","title":"Slow-Fast Architecture for Video Multi-Modal Large Language Models","date":"2025-04-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"7606525af238fb64","mcp_get_code":{"code_sha256":"7606525af238fb64"}},{"arxiv_id":"2504.00502","paper":null,"title":"arXiv:2504.00502","date":null,"month_inferred_from_arxiv_id":"2025-04","title_source":null,"repo":"icip-cas/ShortV","path":"llava/model/llava_arch.py","file_url":"https://github.com/icip-cas/ShortV/blob/HEAD/llava/model/llava_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"55c32993da87759b","mcp_get_code":{"code_sha256":"55c32993da87759b"}},{"arxiv_id":"2503.23463","paper":"/paper/opendrivevla-towards-end-to-end-autonomous","title":"OpenDriveVLA: Towards End-to-end Autonomous Driving with Large Vision Language Action Model","date":"2025-03-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"DriveVLA/OpenDriveVLA","path":"llava/model/llava_arch.py","file_url":"https://github.com/DriveVLA/OpenDriveVLA/blob/HEAD/llava/model/llava_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"7606525af238fb64","mcp_get_code":{"code_sha256":"7606525af238fb64"}},{"arxiv_id":"2503.16013","paper":"/paper/graspcot-integrating-physical-property","title":"GraspCoT: Integrating Physical Property Reasoning for 6-DoF Grasping under Flexible Language Instructions","date":"2025-03-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cxmomo/GraspCoT","path":"llava/model/llava_arch.py","file_url":"https://github.com/cxmomo/GraspCoT/blob/HEAD/llava/model/llava_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"7606525af238fb64","mcp_get_code":{"code_sha256":"7606525af238fb64"}},{"arxiv_id":"2503.15621","paper":"/paper/llava-more-a-comparative-study-of-llms-and","title":"LLaVA-MORE: A Comparative Study of LLMs and Visual Backbones for Enhanced Visual Instruction Tuning","date":"2025-03-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"55c32993da87759b","mcp_get_code":{"code_sha256":"55c32993da87759b"}},{"arxiv_id":"2503.10742","paper":"/paper/keyframe-oriented-vision-token-pruning","title":"Keyframe-oriented Vision Token Pruning: Enhancing Efficiency of Large Vision Language Models on Long-Form Video Processing","date":"2025-03-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"1999Lyd/KVTP","path":"llava/model/llava_arch.py","file_url":"https://github.com/1999Lyd/KVTP/blob/HEAD/llava/model/llava_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"7606525af238fb64","mcp_get_code":{"code_sha256":"7606525af238fb64"}},{"arxiv_id":"2503.08689","paper":"/paper/quota-query-oriented-token-assignment-via-cot","title":"QuoTA: Query-oriented Token Assignment via CoT Query Decouple for Long Video Comprehension","date":"2025-03-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"7606525af238fb64","mcp_get_code":{"code_sha256":"7606525af238fb64"}},{"arxiv_id":"2502.13508","paper":"/paper/vlas-vision-language-action-model-with-speech","title":"VLAS: Vision-Language-Action Model With Speech Instructions For Customized Robot Manipulation","date":null,"month_inferred_from_arxiv_id":"2025-02","title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"55c32993da87759b","mcp_get_code":{"code_sha256":"55c32993da87759b"}},{"arxiv_id":"2501.14818","paper":"/paper/eagle-2-building-post-training-data","title":"Eagle 2: Building Post-Training Data Strategies from Scratch for Frontier Vision-Language Models","date":"2025-01-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"7606525af238fb64","mcp_get_code":{"code_sha256":"7606525af238fb64"}},{"arxiv_id":"2412.13871","paper":"/paper/llava-uhd-v2-an-mllm-integrating-high","title":"LLaVA-UHD v2: an MLLM Integrating High-Resolution Feature Pyramid via Hierarchical Window Transformer","date":"2024-12-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thunlp/llava-uhd","path":"llava/model/llava_arch.py","file_url":"https://github.com/thunlp/llava-uhd/blob/HEAD/llava/model/llava_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"7606525af238fb64","mcp_get_code":{"code_sha256":"7606525af238fb64"}},{"arxiv_id":"2412.07689","paper":"/paper/drivemm-all-in-one-large-multimodal-model-for","title":"DriveMM: All-in-One Large Multimodal Model for Autonomous Driving","date":"2024-12-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zhijian11/DriveMM","path":"llava/model/llava_arch.py","file_url":"https://github.com/zhijian11/DriveMM/blob/HEAD/llava/model/llava_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"7606525af238fb64","mcp_get_code":{"code_sha256":"7606525af238fb64"}},{"arxiv_id":"2412.05819","paper":"/paper/cls-token-tells-everything-needed-for","title":"[CLS] Token Tells Everything Needed for Training-free Efficient MLLMs","date":"2024-12-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thu-mig/vtc-cls","path":"llava/model/llava_arch.py","file_url":"https://github.com/thu-mig/vtc-cls/blob/HEAD/llava/model/llava_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"7606525af238fb64","mcp_get_code":{"code_sha256":"7606525af238fb64"}},{"arxiv_id":"2412.04449","paper":"/paper/p-mod-building-mixture-of-depths-mllms-via","title":"p-MoD: Building Mixture-of-Depths MLLMs via Progressive Ratio Decay","date":"2024-12-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mcg-nju/p-mod","path":"llava/model/llava_arch.py","file_url":"https://github.com/mcg-nju/p-mod/blob/HEAD/llava/model/llava_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"55c32993da87759b","mcp_get_code":{"code_sha256":"55c32993da87759b"}},{"arxiv_id":"2412.03248","paper":"/paper/aim-adaptive-inference-of-multi-modal-llms","title":"AIM: Adaptive Inference of Multi-Modal LLMs via Token Merging and Pruning","date":"2024-12-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lavi-lab/aim","path":"llava/model/llava_arch.py","file_url":"https://github.com/lavi-lab/aim/blob/HEAD/llava/model/llava_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"7606525af238fb64","mcp_get_code":{"code_sha256":"7606525af238fb64"}},{"arxiv_id":"2412.02158","paper":"/paper/agri-llava-knowledge-infused-large-multimodal","title":"Agri-LLaVA: Knowledge-Infused Large Multimodal Assistant on Agricultural Pests and Diseases","date":"2024-12-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kki2eve/agri-llava","path":"agri_llava/model/llava_arch.py","file_url":"https://github.com/kki2eve/agri-llava/blob/HEAD/agri_llava/model/llava_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"55c32993da87759b","mcp_get_code":{"code_sha256":"55c32993da87759b"}},{"arxiv_id":"2412.01818","paper":"/paper/cls-attention-is-all-you-need-for-training","title":"Beyond Text-Visual Attention: Exploiting Visual Cues for Effective Token Pruning in VLMs","date":"2024-12-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"theia-4869/fastervlm","path":"llava/model/llava_arch.py","file_url":"https://github.com/theia-4869/fastervlm/blob/HEAD/llava/model/llava_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"55c32993da87759b","mcp_get_code":{"code_sha256":"55c32993da87759b"}},{"arxiv_id":"2412.00876","paper":"/paper/dynamic-llava-efficient-multimodal-large","title":"Dynamic-LLaVA: Efficient Multimodal Large Language Models via Dynamic Vision-language Context Sparsification","date":"2024-12-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Osilly/dynamic_llava","path":"llava/model/dynamic_llava_arch.py","file_url":"https://github.com/Osilly/dynamic_llava/blob/HEAD/llava/model/dynamic_llava_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"55c32993da87759b","mcp_get_code":{"code_sha256":"55c32993da87759b"}},{"arxiv_id":"2411.15024","paper":"/paper/dycoke-dynamic-compression-of-tokens-for-fast","title":"DyCoke: Dynamic Compression of Tokens for Fast Video Large Language Models","date":"2024-11-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kd-tao/dycoke","path":"llava/model/llava_arch.py","file_url":"https://github.com/kd-tao/dycoke/blob/HEAD/llava/model/llava_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"7606525af238fb64","mcp_get_code":{"code_sha256":"7606525af238fb64"}},{"arxiv_id":"2411.14432","paper":"/paper/insight-v-exploring-long-chain-visual","title":"Insight-V: Exploring Long-Chain Visual Reasoning with Multimodal Large Language Models","date":"2024-11-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"7606525af238fb64","mcp_get_code":{"code_sha256":"7606525af238fb64"}},{"arxiv_id":"2411.11706","paper":"/paper/mc-llava-multi-concept-personalized-vision","title":"MC-LLaVA: Multi-Concept Personalized Vision-Language Model","date":"2024-11-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"7606525af238fb64","mcp_get_code":{"code_sha256":"7606525af238fb64"}},{"arxiv_id":"2411.11360","paper":"/paper/ccexpert-advancing-mllm-capability-in-remote","title":"CCExpert: Advancing MLLM Capability in Remote Sensing Change Captioning with Difference-Aware Integration and a Foundational Dataset","date":"2024-11-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"meize0729/ccexpert","path":"llava/model/llava_arch.py","file_url":"https://github.com/meize0729/ccexpert/blob/HEAD/llava/model/llava_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"7606525af238fb64","mcp_get_code":{"code_sha256":"7606525af238fb64"}},{"arxiv_id":"2411.10803","paper":"/paper/multi-stage-vision-token-dropping-towards","title":"Multi-Stage Vision Token Dropping: Towards Efficient Multimodal Large Language Model","date":"2024-11-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"liuting20/mustdrop","path":"llava/model/llava_arch.py","file_url":"https://github.com/liuting20/mustdrop/blob/HEAD/llava/model/llava_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"55c32993da87759b","mcp_get_code":{"code_sha256":"55c32993da87759b"}},{"arxiv_id":"2411.03312","paper":"/paper/inference-optimal-vlms-need-only-one-visual","title":"Inference Optimal VLMs Need Fewer Visual Tokens and More Parameters","date":"2024-11-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"55c32993da87759b","mcp_get_code":{"code_sha256":"55c32993da87759b"}},{"arxiv_id":"2411.00394","paper":"/paper/right-this-way-can-vlms-guide-us-to-see-more","title":"Right this way: Can VLMs Guide Us to See More to Answer Questions?","date":"2024-11-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"55c32993da87759b","mcp_get_code":{"code_sha256":"55c32993da87759b"}},{"arxiv_id":"2410.22313","paper":"/paper/senna-bridging-large-vision-language-models","title":"Senna: Bridging Large Vision-Language Models and End-to-End Autonomous Driving","date":"2024-10-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hustvl/senna","path":"llava/senna/senna_llava_arch.py","file_url":"https://github.com/hustvl/senna/blob/HEAD/llava/senna/senna_llava_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"55c32993da87759b","mcp_get_code":{"code_sha256":"55c32993da87759b"}},{"arxiv_id":"2410.17434","paper":"/paper/longvu-spatiotemporal-adaptive-compression","title":"LongVU: Spatiotemporal Adaptive Compression for Long Video-Language Understanding","date":"2024-10-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Vision-CAIR/LongVU","path":"longvu/cambrian_arch.py","file_url":"https://github.com/Vision-CAIR/LongVU/blob/HEAD/longvu/cambrian_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"83d2977826140fdc","mcp_get_code":{"code_sha256":"83d2977826140fdc"}},{"arxiv_id":"2410.17247","paper":"/paper/pyramiddrop-accelerating-your-large-vision","title":"PyramidDrop: Accelerating Your Large Vision-Language Models via Pyramid Visual Redundancy Reduction","date":"2024-10-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cooperx521/pyramiddrop","path":"llava/model/llava_arch.py","file_url":"https://github.com/cooperx521/pyramiddrop/blob/HEAD/llava/model/llava_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"55c32993da87759b","mcp_get_code":{"code_sha256":"55c32993da87759b"}},{"arxiv_id":"2410.13360","paper":"/paper/remember-retrieve-and-generate-understanding","title":"RAP: Retrieval-Augmented Personalization for Multimodal Large Language Models","date":"2024-10-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"7606525af238fb64","mcp_get_code":{"code_sha256":"7606525af238fb64"}},{"arxiv_id":"2410.09575","paper":"/paper/reconstructive-visual-instruction-tuning","title":"Reconstructive Visual Instruction Tuning","date":"2024-10-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"haochen-wang409/ross","path":"ross/model/ross_arch.py","file_url":"https://github.com/haochen-wang409/ross/blob/HEAD/ross/model/ross_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"55c32993da87759b","mcp_get_code":{"code_sha256":"55c32993da87759b"}},{"arxiv_id":"2410.02080","paper":"/paper/emma-efficient-visual-alignment-in-multi","title":"EMMA: Efficient Visual Alignment in Multi-Modal LLMs","date":"2024-10-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"55c32993da87759b","mcp_get_code":{"code_sha256":"55c32993da87759b"}},{"arxiv_id":"2408.15998","paper":"/paper/eagle-exploring-the-design-space-for","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","date":"2024-08-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"7606525af238fb64","mcp_get_code":{"code_sha256":"7606525af238fb64"}},{"arxiv_id":"2408.10945","paper":"/paper/hired-attention-guided-token-dropping-for","title":"HiRED: Attention-Guided Token Dropping for Efficient Inference of High-Resolution Vision-Language Models","date":"2024-08-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hasanar1f/hired","path":"transformers/src/transformers/models/llava_next/modeling_llava_next.py","file_url":"https://github.com/hasanar1f/hired/blob/HEAD/transformers/src/transformers/models/llava_next/modeling_llava_next.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f5183d69498ae311","mcp_get_code":{"code_sha256":"f5183d69498ae311"}},{"arxiv_id":"2408.00491","paper":"/paper/2408-00491","title":"GalleryGPT: Analyzing Paintings with Large Multimodal Models","date":"2024-08-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"steven640pixel/gallerygpt","path":"llava/model/llava_arch.py","file_url":"https://github.com/steven640pixel/gallerygpt/blob/HEAD/llava/model/llava_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"7606525af238fb64","mcp_get_code":{"code_sha256":"7606525af238fb64"}},{"arxiv_id":"2407.15841","paper":"/paper/slowfast-llava-a-strong-training-free","title":"SlowFast-LLaVA: A Strong Training-Free Baseline for Video Large Language Models","date":"2024-07-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"apple/ml-slowfast-llava","path":"slowfast_llava/llava/model/llava_arch.py","file_url":"https://github.com/apple/ml-slowfast-llava/blob/HEAD/slowfast_llava/llava/model/llava_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"55c32993da87759b","mcp_get_code":{"code_sha256":"55c32993da87759b"}},{"arxiv_id":"2407.07895","paper":"/paper/llava-next-interleave-tackling-multi-image","title":"LLaVA-NeXT-Interleave: Tackling Multi-image, Video, and 3D in Large Multimodal Models","date":"2024-07-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"7606525af238fb64","mcp_get_code":{"code_sha256":"7606525af238fb64"}},{"arxiv_id":"2406.20092","paper":"/paper/llavolta-efficient-multi-modal-models-via","title":"Efficient Large Multi-modal Models via Visual Context Compression","date":"2024-06-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Beckschen/LLaVolta","path":"llava/model/llava_arch.py","file_url":"https://github.com/Beckschen/LLaVolta/blob/HEAD/llava/model/llava_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"7606525af238fb64","mcp_get_code":{"code_sha256":"7606525af238fb64"}},{"arxiv_id":"2406.16860","paper":"/paper/cambrian-1-a-fully-open-vision-centric","title":"Cambrian-1: A Fully Open, Vision-Centric Exploration of Multimodal LLMs","date":"2024-06-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cambrian-mllm/cambrian","path":"cambrian/model/cambrian_arch.py","file_url":"https://github.com/cambrian-mllm/cambrian/blob/HEAD/cambrian/model/cambrian_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"83d2977826140fdc","mcp_get_code":{"code_sha256":"83d2977826140fdc"}},{"arxiv_id":"2406.16562","paper":"/paper/evalalign-evaluating-text-to-image-models","title":"EVALALIGN: Supervised Fine-Tuning Multimodal LLMs with Human-Aligned Data for Evaluating Text-to-Image Models","date":"2024-06-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sais-fuxi/evalalign","path":"evalalign/model/llava_arch.py","file_url":"https://github.com/sais-fuxi/evalalign/blob/HEAD/evalalign/model/llava_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"55c32993da87759b","mcp_get_code":{"code_sha256":"55c32993da87759b"}},{"arxiv_id":"2406.12275","paper":"/paper/voco-llama-towards-vision-compression-with","title":"VoCo-LLaMA: Towards Vision Compression with Large Language Models","date":"2024-06-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"7606525af238fb64","mcp_get_code":{"code_sha256":"7606525af238fb64"}},{"arxiv_id":"2406.11327","paper":"/paper/clawmachine-fetching-visual-tokens-as-an","title":"ClawMachine: Learning to Fetch Visual Tokens for Referential Comprehension","date":"2024-06-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"martian422/ClawMachine","path":"ClawMachine/model/llava_arch.py","file_url":"https://github.com/martian422/ClawMachine/blob/HEAD/ClawMachine/model/llava_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"7606525af238fb64","mcp_get_code":{"code_sha256":"7606525af238fb64"}},{"arxiv_id":"2406.10995","paper":"/paper/concept-skill-transferability-based-data","title":"Concept-skill Transferability-based Data Selection for Large Vision-Language Models","date":"2024-06-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"g-jwlee/coincide_code","path":"COINCIDE_cluster/tinyllava/model/llava_arch.py","file_url":"https://github.com/g-jwlee/coincide_code/blob/HEAD/COINCIDE_cluster/tinyllava/model/llava_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"7606525af238fb64","mcp_get_code":{"code_sha256":"7606525af238fb64"}},{"arxiv_id":"2406.08487","paper":"/paper/beyond-llava-hd-diving-into-high-resolution","title":"Beyond LLaVA-HD: Diving into High-Resolution Large Multimodal Models","date":"2024-06-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"7606525af238fb64","mcp_get_code":{"code_sha256":"7606525af238fb64"}},{"arxiv_id":"2406.02539","paper":"/paper/parrot-multilingual-visual-instruction-tuning","title":"Parrot: Multilingual Visual Instruction Tuning","date":"2024-06-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"AIDC-AI/Parrot","path":"parrot/model/parrot_arch.py","file_url":"https://github.com/AIDC-AI/Parrot/blob/HEAD/parrot/model/parrot_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"7606525af238fb64","mcp_get_code":{"code_sha256":"7606525af238fb64"}},{"arxiv_id":"2405.19315","paper":"/paper/matryoshka-query-transformer-for-large-vision","title":"Matryoshka Query Transformer for Large Vision-Language Models","date":"2024-05-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"gordonhu608/mqt-llava","path":"llava/model/llava_arch.py","file_url":"https://github.com/gordonhu608/mqt-llava/blob/HEAD/llava/model/llava_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"7606525af238fb64","mcp_get_code":{"code_sha256":"7606525af238fb64"}},{"arxiv_id":"2405.15738","paper":"/paper/convllava-hierarchical-backbones-as-visual","title":"ConvLLaVA: Hierarchical Backbones as Visual Encoder for Large Multimodal Models","date":"2024-05-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"alibaba/conv-llava","path":"llava/model/llava_arch.py","file_url":"https://github.com/alibaba/conv-llava/blob/HEAD/llava/model/llava_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"7606525af238fb64","mcp_get_code":{"code_sha256":"7606525af238fb64"}},{"arxiv_id":"2405.07798","paper":"/paper/freeva-offline-mllm-as-training-free-video","title":"FreeVA: Offline MLLM as Training-Free Video Assistant","date":"2024-05-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"whwu95/freeva","path":"llava/model/llava_arch.py","file_url":"https://github.com/whwu95/freeva/blob/HEAD/llava/model/llava_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"7606525af238fb64","mcp_get_code":{"code_sha256":"7606525af238fb64"}},{"arxiv_id":"2405.05803","paper":"/paper/boosting-multimodal-large-language-models","title":"Boosting Multimodal Large Language Models with Visual Tokens Withdrawal for Rapid Inference","date":"2024-05-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lzhxmu/vtw","path":"llava/model/llava_arch.py","file_url":"https://github.com/lzhxmu/vtw/blob/HEAD/llava/model/llava_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"55c32993da87759b","mcp_get_code":{"code_sha256":"55c32993da87759b"}},{"arxiv_id":"2404.13046","paper":"/paper/mova-adapting-mixture-of-vision-experts-to","title":"MoVA: Adapting Mixture of Vision Experts to Multimodal Context","date":"2024-04-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"templex98/mova","path":"mova/model/arch.py","file_url":"https://github.com/templex98/mova/blob/HEAD/mova/model/arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"55c32993da87759b","mcp_get_code":{"code_sha256":"55c32993da87759b"}},{"arxiv_id":"2312.17243","paper":"/paper/unsupervised-universal-image-segmentation","title":"Unsupervised Universal Image Segmentation","date":"2023-12-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dantong88/llarva","path":"llava/model/llava_arch.py","file_url":"https://github.com/dantong88/llarva/blob/HEAD/llava/model/llava_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"55c32993da87759b","mcp_get_code":{"code_sha256":"55c32993da87759b"}},{"arxiv_id":"2310.03744","paper":"/paper/improved-baselines-with-visual-instruction","title":"Improved Baselines with Visual Instruction Tuning","date":"2023-10-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"55c32993da87759b","mcp_get_code":{"code_sha256":"55c32993da87759b"}},{"arxiv_id":"2309.09958","paper":"/paper/an-empirical-study-of-scaling-instruct-tuned","title":"An Empirical Study of Scaling Instruct-Tuned Large Multimodal Models","date":"2023-09-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"55c32993da87759b","mcp_get_code":{"code_sha256":"55c32993da87759b"}},{"arxiv_id":"2304.08485","paper":"/paper/visual-instruction-tuning-1","title":"Visual Instruction Tuning","date":"2023-04-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"haotian-liu/LLaVA","path":"llava/model/llava_arch.py","file_url":"https://github.com/haotian-liu/LLaVA/blob/HEAD/llava/model/llava_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"55c32993da87759b","mcp_get_code":{"code_sha256":"55c32993da87759b"}},{"arxiv_id":"2304.08485","paper":"/paper/visual-instruction-tuning-1","title":"Visual Instruction Tuning","date":"2023-04-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"LLaVA-VL/LLaVA-NeXT","path":"llava/model/llava_arch.py","file_url":"https://github.com/LLaVA-VL/LLaVA-NeXT/blob/HEAD/llava/model/llava_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"7606525af238fb64","mcp_get_code":{"code_sha256":"7606525af238fb64"}},{"arxiv_id":"2107.10833","paper":"/paper/real-esrgan-training-real-world-blind-super","title":"Real-ESRGAN: Training Real-World Blind Super-Resolution with Pure Synthetic Data","date":"2021-07-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sberbank-ai/real-esrgan","path":"RealESRGAN/utils.py","file_url":"https://github.com/sberbank-ai/real-esrgan/blob/HEAD/RealESRGAN/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"ce655e9f8ddda6d8","mcp_get_code":{"code_sha256":"ce655e9f8ddda6d8"}},{"arxiv_id":"Zhong_AIM_Adaptive_Inference_of_Multi-Modal_LLMs_via_Token_Merging_and_ICCV_2025_paper","paper":null,"title":"arXiv:Zhong_AIM_Adaptive_Inference_of_Multi-Modal_LLMs_via_Token_Merging_and_ICCV_2025_paper","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"LaVi-Lab/AIM","path":"llava/model/llava_arch.py","file_url":"https://github.com/LaVi-Lab/AIM/blob/HEAD/llava/model/llava_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"7606525af238fb64","mcp_get_code":{"code_sha256":"7606525af238fb64"}},{"arxiv_id":"Zhang_Beyond_Training_Dynamic_Token_Merging_for_Zero-Shot_Video_Understanding_ICCV_2025_paper","paper":null,"title":"arXiv:Zhang_Beyond_Training_Dynamic_Token_Merging_for_Zero-Shot_Video_Understanding_ICCV_2025_paper","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"Jam1ezhang/DYTO","path":"dyto/llava/model/llava_arch.py","file_url":"https://github.com/Jam1ezhang/DYTO/blob/HEAD/dyto/llava/model/llava_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"55c32993da87759b","mcp_get_code":{"code_sha256":"55c32993da87759b"}},{"arxiv_id":"Xing_Conical_Visual_Concentration_for_Efficient_Large_Vision-Language_Models_CVPR_2025_paper","paper":null,"title":"arXiv:Xing_Conical_Visual_Concentration_for_Efficient_Large_Vision-Language_Models_CVPR_2025_paper","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"Cooperx521/PyramidDrop","path":"llava/model/llava_arch.py","file_url":"https://github.com/Cooperx521/PyramidDrop/blob/HEAD/llava/model/llava_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"55c32993da87759b","mcp_get_code":{"code_sha256":"55c32993da87759b"}},{"arxiv_id":"2025.findings-acl.458","paper":null,"title":"arXiv:2025.findings-acl.458","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"DCDmllm/Align2LLaVA","path":"reward_model/llava/model/llava_arch.py","file_url":"https://github.com/DCDmllm/Align2LLaVA/blob/HEAD/reward_model/llava/model/llava_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"7606525af238fb64","mcp_get_code":{"code_sha256":"7606525af238fb64"}},{"arxiv_id":"2025.findings-acl.433","paper":null,"title":"arXiv:2025.findings-acl.433","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"haon-chen/mmE5","path":"src/vlm_backbone/llava_next/modeling_llava_next.py","file_url":"https://github.com/haon-chen/mmE5/blob/HEAD/src/vlm_backbone/llava_next/modeling_llava_next.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0f1f7661af2b16ac","mcp_get_code":{"code_sha256":"0f1f7661af2b16ac"}},{"arxiv_id":"2025.findings-acl.359","paper":null,"title":"arXiv:2025.findings-acl.359","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"G-JWLee/TAMP","path":"llava/model/llava_arch.py","file_url":"https://github.com/G-JWLee/TAMP/blob/HEAD/llava/model/llava_arch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"7606525af238fb64","mcp_get_code":{"code_sha256":"7606525af238fb64"}}]}