{"url":"/task/image-comprehension","name":"Image Comprehension","slug":"image-comprehension","description_markdown":null,"categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":49,"papers_with_code":25,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":1,"subtasks":0,"parent_tasks":0},"benchmarks":[],"datasets":[{"url":"/dataset/visual7w","name":"Visual7W","full_name":"","num_papers_in_archive":112}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":25,"of":25,"tagged_in_all":49,"items":[{"url":"/paper/internlm-xcomposer-a-vision-language-large","title":"InternLM-XComposer: A Vision-Language Large Model for Advanced Text-image Comprehension and Composition","date":"2023-09-26","arxiv_id":"2309.15112","repositories_listed":3,"syntology":null},{"url":"/paper/enhancing-visual-language-modality-alignment","title":"Enhancing Visual-Language Modality Alignment in Large Vision Language Models via Self-Improvement","date":"2024-05-24","arxiv_id":"2405.15973","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/mini-gemini-mining-the-potential-of-multi","title":"Mini-Gemini: Mining the Potential of Multi-modality Vision Language Models","date":"2024-03-27","arxiv_id":"2403.18814","repositories_listed":2,"syntology":{"n":8,"n_ran":7,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/new-dataset-and-methods-for-fine-grained","title":"New Dataset and Methods for Fine-Grained Compositional Referring Expression Comprehension via Specialist-MLLM Collaboration","date":"2025-02-27","arxiv_id":"2502.20104","repositories_listed":1,"syntology":null},{"url":"/paper/rrhf-v-ranking-responses-to-mitigate","title":"RRHF-V: Ranking Responses to Mitigate Hallucinations in Multimodal Large Language Models with Human Feedback","date":"2025-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/rsunivlm-a-unified-vision-language-model-for","title":"RSUniVLM: A Unified Vision Language Model for Remote Sensing via Granularity-oriented Mixture of Experts","date":"2024-12-07","arxiv_id":"2412.05679","repositories_listed":1,"syntology":null},{"url":"/paper/divot-diffusion-powers-video-tokenizer-for","title":"Divot: Diffusion Powers Video Tokenizer for Comprehension and Generation","date":"2024-12-05","arxiv_id":"2412.04432","repositories_listed":1,"syntology":null},{"url":"/paper/mmgenbench-evaluating-the-limits-of-lmms-from","title":"MMGenBench: Evaluating the Limits of LMMs from the Text-to-Image Generation Perspective","date":"2024-11-21","arxiv_id":"2411.14062","repositories_listed":1,"syntology":null},{"url":"/paper/clic-contrastive-learning-framework-for","title":"CLIC: Contrastive Learning Framework for Unsupervised Image Complexity Representation","date":"2024-11-19","arxiv_id":"2411.12792","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-multimodal-query-representation-via","title":"MIRe: Enhancing Multimodal Queries Representation via Fusion-Free Modality Interaction for Multimodal Retrieval","date":"2024-11-13","arxiv_id":"2411.08334","repositories_listed":1,"syntology":null},{"url":"/paper/streamingbench-assessing-the-gap-for-mllms-to","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","date":"2024-11-06","arxiv_id":"2411.03628","repositories_listed":1,"syntology":null},{"url":"/paper/ftii-bench-a-comprehensive-multimodal","title":"FTII-Bench: A Comprehensive Multimodal Benchmark for Flow Text with Image Insertion","date":"2024-10-16","arxiv_id":"2410.12564","repositories_listed":1,"syntology":null},{"url":"/paper/finecops-ref-a-new-dataset-and-task-for-fine","title":"FineCops-Ref: A new Dataset and Task for Fine-Grained Compositional Referring Expression Comprehension","date":"2024-09-23","arxiv_id":"2409.14750","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_unverified":0,"n_pointer_only":6}},{"url":"/paper/2408-02718","title":"MMIU: Multimodal Multi-image Understanding for Evaluating Large Vision-Language Models","date":"2024-08-05","arxiv_id":"2408.02718","repositories_listed":1,"syntology":null},{"url":"/paper/paying-more-attention-to-image-a-training","title":"Paying More Attention to Image: A Training-Free Method for Alleviating Hallucination in LVLMs","date":"2024-07-31","arxiv_id":"2407.21771","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/internlm-xcomposer-2-5-a-versatile-large","title":"InternLM-XComposer-2.5: A Versatile Large Vision Language Model Supporting Long-Contextual Input and Output","date":"2024-07-03","arxiv_id":"2407.03320","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/vga-vision-gui-assistant-minimizing","title":"VGA: Vision GUI Assistant -- Minimizing Hallucinations through Image-Centric Fine-Tuning","date":"2024-06-20","arxiv_id":"2406.14056","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-large-vision-language-models-with","title":"Enhancing Large Vision Language Models with Self-Training on Image Comprehension","date":"2024-05-30","arxiv_id":"2405.19716","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/advancing-geometric-problem-solving-a","title":"MM-MATH: Advancing Multimodal Math Evaluation with Process Evaluation and Fine-grained Classification","date":"2024-04-07","arxiv_id":"2404.05091","repositories_listed":1,"syntology":null},{"url":"/paper/earthgpt-a-universal-multi-modal-large","title":"EarthGPT: A Universal Multi-modal Large Language Model for Multi-sensor Image Comprehension in Remote Sensing Domain","date":"2024-01-30","arxiv_id":"2401.16822","repositories_listed":1,"syntology":null},{"url":"/paper/cocot-contrastive-chain-of-thought-prompting","title":"CoCoT: Contrastive Chain-of-Thought Prompting for Large Multimodal Models with Multiple Image Inputs","date":"2024-01-05","arxiv_id":"2401.02582","repositories_listed":1,"syntology":null},{"url":"/paper/regionblip-a-unified-multi-modal-pre-training","title":"RegionBLIP: A Unified Multi-modal Pre-training Framework for Holistic and Regional Comprehension","date":"2023-08-03","arxiv_id":"2308.02299","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/journeydb-a-benchmark-for-generative-image","title":"JourneyDB: A Benchmark for Generative Image Understanding","date":"2023-07-03","arxiv_id":"2307.00716","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-open-vocabulary-universal-image-1","title":"Hierarchical Open-vocabulary Universal Image Segmentation","date":"2023-07-03","arxiv_id":"2307.00764","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/artgpt-4-artistic-vision-language","title":"ArtGPT-4: Towards Artistic-understanding Large Vision-Language Models with Enhanced Adapter","date":"2023-05-12","arxiv_id":"2305.07490","repositories_listed":1,"syntology":null}],"syntology_records":8,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}