{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/find-closest-aspect-ratio","entry":"find_closest_aspect_ratio","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":45,"n_papers_ran":45,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":2,"n_samples_ran":2,"n_samples_fingerprinted":0,"n_places":45,"n_places_pointer_only":26,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":1,"ran":1,"unverified":0},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2606.16158","paper":"/paper/arxiv-2606-16158","title":"Focus When Necessary: Adaptive Routing and Collaborative Grounding for Training-Free Visual Grounding","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"TencentBAC/LazyMCoT","path":"Internvl/utiles_internvl.py","file_url":"https://github.com/TencentBAC/LazyMCoT/blob/HEAD/Internvl/utiles_internvl.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bd77f5f8067f18e9","mcp_get_code":{"code_sha256":"bd77f5f8067f18e9"}},{"arxiv_id":"2606.07289","paper":"/paper/arxiv-2606-07289","title":"Closed-Form Spectral Regularization for Multi-Task Model Merging","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"WalkerWorldPeace/MLLMerging","path":"InternVL/internvl_chat/model_merging.py","file_url":"https://github.com/WalkerWorldPeace/MLLMerging/blob/HEAD/InternVL/internvl_chat/model_merging.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bd77f5f8067f18e9","mcp_get_code":{"code_sha256":"bd77f5f8067f18e9"}},{"arxiv_id":"2605.31351","paper":"/paper/arxiv-2605-31351","title":"VIABLE: A Visually Impaired Assistance Benchmark for VLM-as-a-Judge Evaluation","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"YiyiyiZhao/VIABLE","path":"viable/run_effectiveness/judge_infer_direct/utils_effectiveness/inference_internvl.py","file_url":"https://github.com/YiyiyiZhao/VIABLE/blob/HEAD/viable/run_effectiveness/judge_infer_direct/utils_effectiveness/inference_internvl.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bd77f5f8067f18e9","mcp_get_code":{"code_sha256":"bd77f5f8067f18e9"}},{"arxiv_id":"2605.00273","paper":"/paper/arxiv-2605-00273","title":"When Do Diffusion Models Learn to Generate Multiple Objects?","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"eugene6923/MOSAIC","path":"mosaic/comfort_utils/model_utils/intern_vl.py","file_url":"https://github.com/eugene6923/MOSAIC/blob/HEAD/mosaic/comfort_utils/model_utils/intern_vl.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bd77f5f8067f18e9","mcp_get_code":{"code_sha256":"bd77f5f8067f18e9"}},{"arxiv_id":"2603.00515","paper":"/paper/arxiv-2603-00515","title":"MLLM-4D: Towards Visual-based Spatial-Temporal Intelligence","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"GVCLab/MLLM-4D","path":"evaluation/model_inference/internvideo2_5.py","file_url":"https://github.com/GVCLab/MLLM-4D/blob/HEAD/evaluation/model_inference/internvideo2_5.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bd77f5f8067f18e9","mcp_get_code":{"code_sha256":"bd77f5f8067f18e9"}},{"arxiv_id":"2511.10648","paper":"/paper/arxiv-2511-10648","title":"Enhancing the Outcome Reward-based RL Training of MLLMs with Self-Consistency Sampling","date":null,"month_inferred_from_arxiv_id":"2025-11","title_source":"syntology","repo":"GenuineWWD/SCS","path":"evaluation/eval_internvl_m3cot.py","file_url":"https://github.com/GenuineWWD/SCS/blob/HEAD/evaluation/eval_internvl_m3cot.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bd77f5f8067f18e9","mcp_get_code":{"code_sha256":"bd77f5f8067f18e9"}},{"arxiv_id":"2507.00790","paper":"/paper/ld-rps-zero-shot-unified-image-restoration","title":"LD-RPS: Zero-Shot Unified Image Restoration via Latent Diffusion Recurrent Posterior Sampling","date":"2025-07-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"AMAP-ML/LD-RPS","path":"get_prompts.py","file_url":"https://github.com/AMAP-ML/LD-RPS/blob/HEAD/get_prompts.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bd77f5f8067f18e9","mcp_get_code":{"code_sha256":"bd77f5f8067f18e9"}},{"arxiv_id":"2505.20148","paper":"/paper/mineanybuild-benchmarking-spatial-planning","title":"MineAnyBuild: Benchmarking Spatial Planning for Open-world AI Agents","date":"2025-05-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"code_sha256_prefix":"bd77f5f8067f18e9","mcp_get_code":{"code_sha256":"bd77f5f8067f18e9"}},{"arxiv_id":"2505.19892","paper":"/paper/unifying-multimodal-large-language-model","title":"Unifying Multimodal Large Language Model Capabilities and Modalities via Model Merging","date":"2025-05-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"code_sha256_prefix":"bd77f5f8067f18e9","mcp_get_code":{"code_sha256":"bd77f5f8067f18e9"}},{"arxiv_id":"2505.17609","paper":"/paper/decoupled-visual-interpretation-and","title":"Decoupled Visual Interpretation and Linguistic Reasoning for Math Problem Solving","date":"2025-05-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"code_sha256_prefix":"bd77f5f8067f18e9","mcp_get_code":{"code_sha256":"bd77f5f8067f18e9"}},{"arxiv_id":"2504.15485","paper":null,"title":"arXiv:2504.15485","date":null,"month_inferred_from_arxiv_id":"2025-04","title_source":null,"repo":"atinpothiraj/CAPTURe","path":"occluded_scripts/intern.py","file_url":"https://github.com/atinpothiraj/CAPTURe/blob/HEAD/occluded_scripts/intern.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bd77f5f8067f18e9","mcp_get_code":{"code_sha256":"bd77f5f8067f18e9"}},{"arxiv_id":"2504.10465","paper":"/paper/pixel-sail-single-transformer-for-pixel","title":"Pixel-SAIL: Single Transformer For Pixel-Grounded Understanding","date":"2025-04-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"magic-research/Sa2VA","path":"projects/sa2va/models/utils.py","file_url":"https://github.com/magic-research/Sa2VA/blob/HEAD/projects/sa2va/models/utils.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bd77f5f8067f18e9","mcp_get_code":{"code_sha256":"bd77f5f8067f18e9"}},{"arxiv_id":"2503.15234","paper":"/paper/coe-chain-of-explanation-via-automatic-visual","title":"CoE: Chain-of-Explanation via Automatic Visual Concept Circuit Description and Polysemanticity Quantification","date":"2025-03-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"YuWLong666/CoE","path":"models/internvl.py","file_url":"https://github.com/YuWLong666/CoE/blob/HEAD/models/internvl.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"bd77f5f8067f18e9","mcp_get_code":{"code_sha256":"bd77f5f8067f18e9"}},{"arxiv_id":"2503.12769","paper":"/paper/vispeak-visual-instruction-feedback-in","title":"ViSpeak: Visual Instruction Feedback in Streaming Videos","date":"2025-03-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thunlp-mt/streamingbench","path":"src/model/InternVL.py","file_url":"https://github.com/thunlp-mt/streamingbench/blob/HEAD/src/model/InternVL.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bd77f5f8067f18e9","mcp_get_code":{"code_sha256":"bd77f5f8067f18e9"}},{"arxiv_id":"2502.15969","paper":"/paper/forgotten-polygons-multimodal-large-language","title":"Forgotten Polygons: Multimodal Large Language Models are Shape-Blind","date":"2025-02-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"code_sha256_prefix":"bd77f5f8067f18e9","mcp_get_code":{"code_sha256":"bd77f5f8067f18e9"}},{"arxiv_id":"2502.14882","paper":"/paper/calibquant-1-bit-kv-cache-quantization-for","title":"CalibQuant: 1-Bit KV Cache Quantization for Multimodal LLMs","date":"2025-02-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"code_sha256_prefix":"bd77f5f8067f18e9","mcp_get_code":{"code_sha256":"bd77f5f8067f18e9"}},{"arxiv_id":"2501.05510","paper":"/paper/ovo-bench-how-far-is-your-video-llms-from","title":"OVO-Bench: How Far is Your Video-LLMs from Real-World Online Video Understanding?","date":"2025-01-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"JoeLeelyf/OVO-Bench","path":"models/InternVL2.py","file_url":"https://github.com/JoeLeelyf/OVO-Bench/blob/HEAD/models/InternVL2.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bd77f5f8067f18e9","mcp_get_code":{"code_sha256":"bd77f5f8067f18e9"}},{"arxiv_id":"2412.21080","paper":"/paper/vinci-a-real-time-embodied-smart-assistant","title":"Vinci: A Real-time Embodied Smart Assistant based on Egocentric Vision-Language Model","date":"2024-12-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"code_sha256_prefix":"bd77f5f8067f18e9","mcp_get_code":{"code_sha256":"bd77f5f8067f18e9"}},{"arxiv_id":"2412.12693","paper":"/paper/sphere-a-hierarchical-evaluation-on-spatial","title":"SPHERE: A Hierarchical Evaluation on Spatial Perception and Reasoning for Vision-Language Models","date":"2024-12-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zwenyu/SPHERE-VLM","path":"models/vision_language_models/intern_vl2_5.py","file_url":"https://github.com/zwenyu/SPHERE-VLM/blob/HEAD/models/vision_language_models/intern_vl2_5.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bd77f5f8067f18e9","mcp_get_code":{"code_sha256":"bd77f5f8067f18e9"}},{"arxiv_id":"2412.06171","paper":"/paper/holmes-vau-towards-long-term-video-anomaly","title":"Holmes-VAU: Towards Long-term Video Anomaly Understanding at Any Granularity","date":"2024-12-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pipixin321/holmesvau","path":"holmesvau/internvl_utils.py","file_url":"https://github.com/pipixin321/holmesvau/blob/HEAD/holmesvau/internvl_utils.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bd77f5f8067f18e9","mcp_get_code":{"code_sha256":"bd77f5f8067f18e9"}},{"arxiv_id":"2412.05185","paper":"/paper/linvt-empower-your-image-level-large-language","title":"LinVT: Empower Your Image-level Large Language Model to Understand Videos","date":"2024-12-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"gls0425/linvt","path":"streamlit_demo/model_worker.py","file_url":"https://github.com/gls0425/linvt/blob/HEAD/streamlit_demo/model_worker.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bd77f5f8067f18e9","mcp_get_code":{"code_sha256":"bd77f5f8067f18e9"}},{"arxiv_id":"2411.11496","paper":"/paper/safe-safe-unsafe-exploring-how-safe-images","title":"Safe + Safe = Unsafe? Exploring How Safe Images Can Be Exploited to Jailbreak Large Vision-Language Models","date":"2024-11-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"gzcch/safety_snowball_agent","path":"InternVL_assitant.py","file_url":"https://github.com/gzcch/safety_snowball_agent/blob/HEAD/InternVL_assitant.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bd77f5f8067f18e9","mcp_get_code":{"code_sha256":"bd77f5f8067f18e9"}},{"arxiv_id":"2411.03823","paper":"/paper/both-text-and-images-leaked-a-systematic","title":"Both Text and Images Leaked! A Systematic Analysis of Multimodal LLM Data Contamination","date":"2024-11-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"MLLM-Data-Contamination/MM-Detect","path":"mm_detect/mllms/internvl2.py","file_url":"https://github.com/MLLM-Data-Contamination/MM-Detect/blob/HEAD/mm_detect/mllms/internvl2.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bd77f5f8067f18e9","mcp_get_code":{"code_sha256":"bd77f5f8067f18e9"}},{"arxiv_id":"2410.17885","paper":"/paper/r-cot-reverse-chain-of-thought-problem","title":"R-CoT: Reverse Chain-of-Thought Problem Generation for Geometric Reasoning in Large Multimodal Models","date":"2024-10-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dle666/r-cot","path":"GeoQA_test/model_vqa_rcot2b.py","file_url":"https://github.com/dle666/r-cot/blob/HEAD/GeoQA_test/model_vqa_rcot2b.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bd77f5f8067f18e9","mcp_get_code":{"code_sha256":"bd77f5f8067f18e9"}},{"arxiv_id":"2410.17385","paper":"/paper/do-vision-language-models-represent-space-and","title":"Do Vision-Language Models Represent Space and How? Evaluating Spatial Frame of Reference Under Ambiguities","date":"2024-10-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sled-group/COMFORT","path":"comfort_utils/model_utils/intern_vl.py","file_url":"https://github.com/sled-group/COMFORT/blob/HEAD/comfort_utils/model_utils/intern_vl.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bd77f5f8067f18e9","mcp_get_code":{"code_sha256":"bd77f5f8067f18e9"}},{"arxiv_id":"2410.14179","paper":"/paper/multichartqa-benchmarking-vision-language","title":"MultiChartQA: Benchmarking Vision-Language Models on Multi-Chart Problems","date":"2024-10-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zivenzhu/multi-chart-qa","path":"code/evaluate_internvl15.py","file_url":"https://github.com/zivenzhu/multi-chart-qa/blob/HEAD/code/evaluate_internvl15.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bd77f5f8067f18e9","mcp_get_code":{"code_sha256":"bd77f5f8067f18e9"}},{"arxiv_id":"2410.09453","paper":"/paper/mmad-the-first-ever-comprehensive-benchmark","title":"MMAD: The First-Ever Comprehensive Benchmark for Multimodal Large Language Models in Industrial Anomaly Detection","date":"2024-10-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"code_sha256_prefix":"bd77f5f8067f18e9","mcp_get_code":{"code_sha256":"bd77f5f8067f18e9"}},{"arxiv_id":"2410.01733","paper":"/paper/visual-perception-in-text-strings","title":"Visual Perception in Text Strings","date":"2024-10-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"JiaQiSJTU/VisionInText","path":"src/evaluation_mm.py","file_url":"https://github.com/JiaQiSJTU/VisionInText/blob/HEAD/src/evaluation_mm.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bd77f5f8067f18e9","mcp_get_code":{"code_sha256":"bd77f5f8067f18e9"}},{"arxiv_id":"2408.08693","paper":"/paper/med-pmc-medical-personalized-multi-modal","title":"Med-PMC: Medical Personalized Multi-modal Consultation with a Proactive Ask-First-Observe-Next Paradigm","date":"2024-08-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"liuhc0428/med-pmc","path":"src/models/InternVL.py","file_url":"https://github.com/liuhc0428/med-pmc/blob/HEAD/src/models/InternVL.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bd77f5f8067f18e9","mcp_get_code":{"code_sha256":"bd77f5f8067f18e9"}},{"arxiv_id":"2407.15240","paper":"/paper/bigbench-a-unified-benchmark-for-social-bias","title":"BIGbench: A Unified Benchmark for Evaluating Multi-dimensional Social Biases in Text-to-Image Models","date":"2024-07-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bigbench2024/bigbench2024","path":"benchmark/internViT_pkg/internvl_detection.py","file_url":"https://github.com/bigbench2024/bigbench2024/blob/HEAD/benchmark/internViT_pkg/internvl_detection.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"GPL-3.0","inline_ok":false,"code_sha256_prefix":"d0487594e4d5c510","mcp_get_code":{"code_sha256":"d0487594e4d5c510"}},{"arxiv_id":"2407.04903","paper":"/paper/mmsci-a-multimodal-multi-discipline-dataset","title":"MMSci: A Dataset for Graduate-Level Multi-Discipline Multimodal Scientific Understanding","date":"2024-07-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"leezekun/mmsci","path":"mmsci-exps/model_loader.py","file_url":"https://github.com/leezekun/mmsci/blob/HEAD/mmsci-exps/model_loader.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bd77f5f8067f18e9","mcp_get_code":{"code_sha256":"bd77f5f8067f18e9"}},{"arxiv_id":"2407.03320","paper":"/paper/internlm-xcomposer-2-5-a-versatile-large","title":"InternLM-XComposer-2.5: A Versatile Large Vision Language Model Supporting Long-Contextual Input and Output","date":"2024-07-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"code_sha256_prefix":"bd77f5f8067f18e9","mcp_get_code":{"code_sha256":"bd77f5f8067f18e9"}},{"arxiv_id":"2407.01523","paper":"/paper/mmlongbench-doc-benchmarking-long-context","title":"MMLongBench-Doc: Benchmarking Long-context Document Understanding with Visualizations","date":"2024-07-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mayubo2333/mmlongbench-doc","path":"models/internvl_chat.py","file_url":"https://github.com/mayubo2333/mmlongbench-doc/blob/HEAD/models/internvl_chat.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bd77f5f8067f18e9","mcp_get_code":{"code_sha256":"bd77f5f8067f18e9"}},{"arxiv_id":"2406.11833","paper":"/paper/mmdu-a-multi-turn-multi-image-dialog","title":"MMDU: A Multi-Turn Multi-Image Dialog Understanding Benchmark and Instruction-Tuning Dataset for LVLMs","date":"2024-06-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"liuziyu77/mmdu","path":"model_generation/InternVL_chat_gen_ans.py","file_url":"https://github.com/liuziyu77/mmdu/blob/HEAD/model_generation/InternVL_chat_gen_ans.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bd77f5f8067f18e9","mcp_get_code":{"code_sha256":"bd77f5f8067f18e9"}},{"arxiv_id":"2406.11794","paper":"/paper/datacomp-lm-in-search-of-the-next-generation","title":"DataComp-LM: In search of the next generation of training sets for language models","date":"2024-06-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jieyuz2/taskmeanything","path":"tma/models/qa_model/imageqa_model.py","file_url":"https://github.com/jieyuz2/taskmeanything/blob/HEAD/tma/models/qa_model/imageqa_model.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bd77f5f8067f18e9","mcp_get_code":{"code_sha256":"bd77f5f8067f18e9"}},{"arxiv_id":"2406.07230","paper":"/paper/needle-in-a-multimodal-haystack","title":"Needle In A Multimodal Haystack","date":"2024-06-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"OpenGVLab/MM-NIAH","path":"eval_internvl.py","file_url":"https://github.com/OpenGVLab/MM-NIAH/blob/HEAD/eval_internvl.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bd77f5f8067f18e9","mcp_get_code":{"code_sha256":"bd77f5f8067f18e9"}},{"arxiv_id":"2406.06462","paper":"/paper/vcr-visual-caption-restoration","title":"VCR: A Task for Pixel-Level Complex Reasoning in Vision Language Models via Restoring Occluded Text","date":"2024-06-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tianyu-z/vcr","path":"src/evaluation/utils.py","file_url":"https://github.com/tianyu-z/vcr/blob/HEAD/src/evaluation/utils.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"CC-BY-SA-4.0","inline_ok":false,"code_sha256_prefix":"bd77f5f8067f18e9","mcp_get_code":{"code_sha256":"bd77f5f8067f18e9"}},{"arxiv_id":"2405.14156","paper":"/paper/unveiling-the-tapestry-of-consistency-in","title":"Unveiling the Tapestry of Consistency in Large Vision-Language Models","date":"2024-05-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"foundation-multimodal-models/conbench","path":"eval/InternVL-Chat-V1-5-26B.py","file_url":"https://github.com/foundation-multimodal-models/conbench/blob/HEAD/eval/InternVL-Chat-V1-5-26B.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bd77f5f8067f18e9","mcp_get_code":{"code_sha256":"bd77f5f8067f18e9"}},{"arxiv_id":"2403.15952","paper":"/paper/illusionvqa-a-challenging-optical-illusion","title":"IllusionVQA: A Challenging Optical Illusion Dataset for Vision Language Models","date":"2024-03-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"csebuetnlp/illusionvqa","path":"inference_code/open_source/internvlm_inference.py","file_url":"https://github.com/csebuetnlp/illusionvqa/blob/HEAD/inference_code/open_source/internvlm_inference.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bd77f5f8067f18e9","mcp_get_code":{"code_sha256":"bd77f5f8067f18e9"}},{"arxiv_id":"2312.13271","paper":"/paper/repaint123-fast-and-high-quality-one-image-to","title":"Repaint123: Fast and High-quality One Image to 3D Generation with Progressive Controllable 2D Repainting","date":"2023-12-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"junwuzhang19/repaint123","path":"main2.py","file_url":"https://github.com/junwuzhang19/repaint123/blob/HEAD/main2.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bd77f5f8067f18e9","mcp_get_code":{"code_sha256":"bd77f5f8067f18e9"}},{"arxiv_id":"1910.01108","paper":"/paper/distilbert-a-distilled-version-of-bert","title":"DistilBERT, a distilled version of BERT: smaller, faster, cheaper and lighter","date":"2019-10-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"reycn/multi-modal-scale","path":"script/3.evaluation.batch.py","file_url":"https://github.com/reycn/multi-modal-scale/blob/HEAD/script/3.evaluation.batch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bd77f5f8067f18e9","mcp_get_code":{"code_sha256":"bd77f5f8067f18e9"}},{"arxiv_id":"Zhang_Holmes-VAU_Towards_Long-term_Video_Anomaly_Understanding_at_Any_Granularity_CVPR_2025_paper","paper":null,"title":"arXiv:Zhang_Holmes-VAU_Towards_Long-term_Video_Anomaly_Understanding_at_Any_Granularity_CVPR_2025_paper","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"pipixin321/HolmesVAU","path":"holmesvau/internvl_utils.py","file_url":"https://github.com/pipixin321/HolmesVAU/blob/HEAD/holmesvau/internvl_utils.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bd77f5f8067f18e9","mcp_get_code":{"code_sha256":"bd77f5f8067f18e9"}},{"arxiv_id":"Yang_PVC_Progressive_Visual_Token_Compression_for_Unified_Image_and_Video_CVPR_2025_paper","paper":null,"title":"arXiv:Yang_PVC_Progressive_Visual_Token_Compression_for_Unified_Image_and_Video_CVPR_2025_paper","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"OpenGVLab/PVC","path":"utils/preprocess.py","file_url":"https://github.com/OpenGVLab/PVC/blob/HEAD/utils/preprocess.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bd77f5f8067f18e9","mcp_get_code":{"code_sha256":"bd77f5f8067f18e9"}},{"arxiv_id":"Guo_Integrating_Visual_Interpretation_and_Linguistic_Reasoning_for_Geometric_Problem_Solving_ICCV_2025_paper","paper":null,"title":"arXiv:Guo_Integrating_Visual_Interpretation_and_Linguistic_Reasoning_for_Geometric_Problem_Solving_ICCV_2025_paper","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"guozix/DVLR","path":"eval/MathVerse/evaluation/generate_response_geo_text_qwen_2.py","file_url":"https://github.com/guozix/DVLR/blob/HEAD/eval/MathVerse/evaluation/generate_response_geo_text_qwen_2.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bd77f5f8067f18e9","mcp_get_code":{"code_sha256":"bd77f5f8067f18e9"}},{"arxiv_id":"2025.acl-long.663","paper":null,"title":"arXiv:2025.acl-long.663","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"BlueZeros/ReflecTool","path":"reflectool/models/InternVLChat.py","file_url":"https://github.com/BlueZeros/ReflecTool/blob/HEAD/reflectool/models/InternVLChat.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bd77f5f8067f18e9","mcp_get_code":{"code_sha256":"bd77f5f8067f18e9"}}]}