{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/auto-pad-images","entry":"auto_pad_images","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":20,"n_papers_ran":0,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":1,"n_samples_ran":0,"n_samples_fingerprinted":0,"n_places":20,"n_places_pointer_only":2,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":0,"ran":0,"unverified":1},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2604.13565","paper":"/paper/arxiv-2604-13565","title":"UHR-BAT: Budget-Aware Token Compression Vision-Language model for Ultra-High-Resolution Remote Sensing","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"Yunkaidang/UHR","path":"with-SAM/longva/longva/mm_utils.py","file_url":"https://github.com/Yunkaidang/UHR/blob/HEAD/with-SAM/longva/longva/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"3c8823948724e269","mcp_get_code":{"code_sha256":"3c8823948724e269"}},{"arxiv_id":"2601.17868","paper":"/paper/arxiv-2601-17868","title":"VidLaDA: Bidirectional Diffusion Large Language Models for Efficient Video Understanding","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"ziHoHe/VidLaDA","path":"train/llava/mm_utils.py","file_url":"https://github.com/ziHoHe/VidLaDA/blob/HEAD/train/llava/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"3c8823948724e269","mcp_get_code":{"code_sha256":"3c8823948724e269"}},{"arxiv_id":"2601.10710","paper":"/paper/arxiv-2601-10710","title":"Cross-Layer Injection for Deep Vision-Language Fusion","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"codefuse-ai/CLI","path":"llava/mm_utils.py","file_url":"https://github.com/codefuse-ai/CLI/blob/HEAD/llava/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"3c8823948724e269","mcp_get_code":{"code_sha256":"3c8823948724e269"}},{"arxiv_id":"2506.21862","paper":"/paper/llava-scissor-token-compression-with-semantic","title":"LLaVA-Scissor: Token Compression with Semantic Connected Components for Video LLMs","date":"2025-06-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"HumanMLLM/LLaVA-Scissor","path":"llava/mm_utils.py","file_url":"https://github.com/HumanMLLM/LLaVA-Scissor/blob/HEAD/llava/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"3c8823948724e269","mcp_get_code":{"code_sha256":"3c8823948724e269"}},{"arxiv_id":"2505.24625","paper":"/paper/learning-from-videos-for-3d-world-enhancing","title":"Learning from Videos for 3D World: Enhancing MLLMs with 3D Vision Geometry Priors","date":"2025-05-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"LaVi-Lab/Video-3D-LLM","path":"llava/mm_utils.py","file_url":"https://github.com/LaVi-Lab/Video-3D-LLM/blob/HEAD/llava/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"3c8823948724e269","mcp_get_code":{"code_sha256":"3c8823948724e269"}},{"arxiv_id":"2505.16839","paper":"/paper/lavida-a-large-diffusion-language-model-for","title":"LaViDa: A Large Diffusion Language Model for Multimodal Understanding","date":"2025-05-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jacklishufan/lavida","path":"llava/mm_utils.py","file_url":"https://github.com/jacklishufan/lavida/blob/HEAD/llava/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"3c8823948724e269","mcp_get_code":{"code_sha256":"3c8823948724e269"}},{"arxiv_id":"2505.14640","paper":"/paper/videoeval-pro-robust-and-realistic-long-video","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","date":"2025-05-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"opengvlab/videochat-flash","path":"llava-train_videochat/llava/mm_utils.py","file_url":"https://github.com/opengvlab/videochat-flash/blob/HEAD/llava-train_videochat/llava/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3c8823948724e269","mcp_get_code":{"code_sha256":"3c8823948724e269"}},{"arxiv_id":"2504.11455","paper":"/paper/simplear-pushing-the-frontier-of","title":"SimpleAR: Pushing the Frontier of Autoregressive Visual Generation through Pretraining, SFT, and RL","date":"2025-04-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wdrink/simplear","path":"simpar/mm_utils.py","file_url":"https://github.com/wdrink/simplear/blob/HEAD/simpar/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3c8823948724e269","mcp_get_code":{"code_sha256":"3c8823948724e269"}},{"arxiv_id":"2503.23463","paper":"/paper/opendrivevla-towards-end-to-end-autonomous","title":"OpenDriveVLA: Towards End-to-end Autonomous Driving with Large Vision Language Action Model","date":"2025-03-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"DriveVLA/OpenDriveVLA","path":"llava/mm_utils.py","file_url":"https://github.com/DriveVLA/OpenDriveVLA/blob/HEAD/llava/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"3c8823948724e269","mcp_get_code":{"code_sha256":"3c8823948724e269"}},{"arxiv_id":"2503.10742","paper":"/paper/keyframe-oriented-vision-token-pruning","title":"Keyframe-oriented Vision Token Pruning: Enhancing Efficiency of Large Vision Language Models on Long-Form Video Processing","date":"2025-03-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"1999Lyd/KVTP","path":"llava/mm_utils.py","file_url":"https://github.com/1999Lyd/KVTP/blob/HEAD/llava/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3c8823948724e269","mcp_get_code":{"code_sha256":"3c8823948724e269"}},{"arxiv_id":"2412.13871","paper":"/paper/llava-uhd-v2-an-mllm-integrating-high","title":"LLaVA-UHD v2: an MLLM Integrating High-Resolution Feature Pyramid via Hierarchical Window Transformer","date":"2024-12-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thunlp/llava-uhd","path":"llava/mm_utils.py","file_url":"https://github.com/thunlp/llava-uhd/blob/HEAD/llava/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"3c8823948724e269","mcp_get_code":{"code_sha256":"3c8823948724e269"}},{"arxiv_id":"2412.09501","paper":"/paper/lyra-an-efficient-and-speech-centric","title":"Lyra: An Efficient and Speech-Centric Framework for Omni-Cognition","date":"2024-12-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dvlab-research/Lyra","path":"lyra/mm_utils.py","file_url":"https://github.com/dvlab-research/Lyra/blob/HEAD/lyra/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"3c8823948724e269","mcp_get_code":{"code_sha256":"3c8823948724e269"}},{"arxiv_id":"2412.07689","paper":"/paper/drivemm-all-in-one-large-multimodal-model-for","title":"DriveMM: All-in-One Large Multimodal Model for Autonomous Driving","date":"2024-12-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zhijian11/DriveMM","path":"llava/mm_utils.py","file_url":"https://github.com/zhijian11/DriveMM/blob/HEAD/llava/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"3c8823948724e269","mcp_get_code":{"code_sha256":"3c8823948724e269"}},{"arxiv_id":"2412.03248","paper":"/paper/aim-adaptive-inference-of-multi-modal-llms","title":"AIM: Adaptive Inference of Multi-Modal LLMs via Token Merging and Pruning","date":"2024-12-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lavi-lab/aim","path":"llava/mm_utils.py","file_url":"https://github.com/lavi-lab/aim/blob/HEAD/llava/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"3c8823948724e269","mcp_get_code":{"code_sha256":"3c8823948724e269"}},{"arxiv_id":"2411.15024","paper":"/paper/dycoke-dynamic-compression-of-tokens-for-fast","title":"DyCoke: Dynamic Compression of Tokens for Fast Video Large Language Models","date":"2024-11-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kd-tao/dycoke","path":"llava/mm_utils.py","file_url":"https://github.com/kd-tao/dycoke/blob/HEAD/llava/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"3c8823948724e269","mcp_get_code":{"code_sha256":"3c8823948724e269"}},{"arxiv_id":"2411.14432","paper":"/paper/insight-v-exploring-long-chain-visual","title":"Insight-V: Exploring Long-Chain Visual Reasoning with Multimodal Large Language Models","date":"2024-11-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dongyh20/insight-v","path":"llava/mm_utils.py","file_url":"https://github.com/dongyh20/insight-v/blob/HEAD/llava/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"3c8823948724e269","mcp_get_code":{"code_sha256":"3c8823948724e269"}},{"arxiv_id":"2411.11360","paper":"/paper/ccexpert-advancing-mllm-capability-in-remote","title":"CCExpert: Advancing MLLM Capability in Remote Sensing Change Captioning with Difference-Aware Integration and a Foundational Dataset","date":"2024-11-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"meize0729/ccexpert","path":"llava/mm_utils.py","file_url":"https://github.com/meize0729/ccexpert/blob/HEAD/llava/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"3c8823948724e269","mcp_get_code":{"code_sha256":"3c8823948724e269"}},{"arxiv_id":"2411.10332","paper":"/paper/number-it-temporal-grounding-videos-like","title":"Number it: Temporal Grounding Videos like Flipping Manga","date":"2024-11-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yongliang-wu/numpro","path":"longva/mm_utils.py","file_url":"https://github.com/yongliang-wu/numpro/blob/HEAD/longva/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3c8823948724e269","mcp_get_code":{"code_sha256":"3c8823948724e269"}},{"arxiv_id":"Zhong_AIM_Adaptive_Inference_of_Multi-Modal_LLMs_via_Token_Merging_and_ICCV_2025_paper","paper":null,"title":"arXiv:Zhong_AIM_Adaptive_Inference_of_Multi-Modal_LLMs_via_Token_Merging_and_ICCV_2025_paper","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"LaVi-Lab/AIM","path":"llava/mm_utils.py","file_url":"https://github.com/LaVi-Lab/AIM/blob/HEAD/llava/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"3c8823948724e269","mcp_get_code":{"code_sha256":"3c8823948724e269"}},{"arxiv_id":"2025.findings-acl.359","paper":null,"title":"arXiv:2025.findings-acl.359","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"G-JWLee/TAMP","path":"llava/mm_utils.py","file_url":"https://github.com/G-JWLee/TAMP/blob/HEAD/llava/mm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"3c8823948724e269","mcp_get_code":{"code_sha256":"3c8823948724e269"}}]}