{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/visual-question-answering/papers/5","list_of":"/task/visual-question-answering","task":"Visual Question Answering (VQA)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":5,"pages_in_order":22,"rows_per_page":100,"rows":[401,500],"of":2167,"counts":{"archive_papers_tagged":2167,"with_a_code_link":1039,"where_syntology_ran_a_sample":359,"not_listed_spam_title":0,"listed":2167,"listed_where_code_ran":359,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":287,"every_run_a_failure_of_syntologys_instrument":72,"listed_with_a_run_with_no_instrument_failure":287,"listed_every_run_a_failure_of_syntologys_instrument":72,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/visual-question-answering","prev":"/task/visual-question-answering/papers/4","next":"/task/visual-question-answering/papers/6","papers":[{"url":"/paper/spiqa-a-dataset-for-multimodal-question","slug":"spiqa-a-dataset-for-multimodal-question","title":"SPIQA: A Dataset for Multimodal Question Answering on Scientific Papers","date":"2024-07-12","arxiv_id":"2407.09413","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/spiqa-a-dataset-for-multimodal-question#ran","syntology_url":"https://syntology.ai/paper/2407.09413","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.09413"}},"official":{"repos":["google/spiqa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-models-understand-layouts","slug":"large-language-models-understand-layouts","title":"Large Language Models Understand Layout","date":"2024-07-08","arxiv_id":"2407.05750","repositories_listed":1,"syntology":null},{"url":"/paper/wsi-vqa-interpreting-whole-slide-images-by","slug":"wsi-vqa-interpreting-whole-slide-images-by","title":"WSI-VQA: Interpreting Whole Slide Images by Generative Visual Question Answering","date":"2024-07-08","arxiv_id":"2407.05603","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/wsi-vqa-interpreting-whole-slide-images-by#ran","syntology_url":"https://syntology.ai/paper/2407.05603","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.05603"}},"official":{"repos":["cpystan/wsi-vqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/clipvqa-video-quality-assessment-via-clip","slug":"clipvqa-video-quality-assessment-via-clip","title":"CLIPVQA:Video Quality Assessment via CLIP","date":"2024-07-06","arxiv_id":"2407.04928","repositories_listed":1,"syntology":null},{"url":"/paper/flowlearn-evaluating-large-vision-language","slug":"flowlearn-evaluating-large-vision-language","title":"FlowLearn: Evaluating Large Vision-Language Models on Flowchart Understanding","date":"2024-07-06","arxiv_id":"2407.05183","repositories_listed":1,"syntology":null},{"url":"/paper/rule-reliable-multimodal-rag-for-factuality","slug":"rule-reliable-multimodal-rag-for-factuality","title":"RULE: Reliable Multimodal RAG for Factuality in Medical Vision Language Models","date":"2024-07-06","arxiv_id":"2407.05131","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rule-reliable-multimodal-rag-for-factuality#ran","syntology_url":"https://syntology.ai/paper/2407.05131","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.05131"}},"official":{"repos":["richard-peng-xia/rule"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/minigpt-med-large-language-model-as-a-general","slug":"minigpt-med-large-language-model-as-a-general","title":"MiniGPT-Med: Large Language Model as a General Interface for Radiology Diagnosis","date":"2024-07-04","arxiv_id":"2407.04106","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/minigpt-med-large-language-model-as-a-general#ran","syntology_url":"https://syntology.ai/paper/2407.04106","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04106"}},"official":{"repos":["vision-cair/minigpt-med"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/what-to-do-if-language-models-disagree-black","slug":"what-to-do-if-language-models-disagree-black","title":"Black-box Model Ensembling for Textual and Visual Question Answering via Information Fusion","date":"2024-07-04","arxiv_id":"2407.12841","repositories_listed":1,"syntology":null},{"url":"/paper/visual-robustness-benchmark-for-visual","slug":"visual-robustness-benchmark-for-visual","title":"Visual Robustness Benchmark for Visual Question Answering (VQA)","date":"2024-07-03","arxiv_id":"2407.03386","repositories_listed":1,"syntology":null},{"url":"/paper/a-bounding-box-is-worth-one-token","slug":"a-bounding-box-is-worth-one-token","title":"A Bounding Box is Worth One Token: Interleaving Layout and Text in a Large Language Model for Document Understanding","date":"2024-07-02","arxiv_id":"2407.01976","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":1,"n_no_contract":4,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 2 honoured, 1 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-bounding-box-is-worth-one-token#ran","syntology_url":"https://syntology.ai/paper/2407.01976","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01976"}},"official":{"repos":["laytextllm/laytextllm"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/m-bench-a-vision-language-benchmark-for","slug":"m-bench-a-vision-language-benchmark-for","title":"μ-Bench: A Vision-Language Benchmark for Microscopy Understanding","date":"2024-07-01","arxiv_id":"2407.01791","repositories_listed":1,"syntology":null},{"url":"/paper/tarsier-recipes-for-training-and-evaluating-1","slug":"tarsier-recipes-for-training-and-evaluating-1","title":"Tarsier: Recipes for Training and Evaluating Large Video Description Models","date":"2024-06-30","arxiv_id":"2407.00634","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tarsier-recipes-for-training-and-evaluating-1#ran","syntology_url":"https://syntology.ai/paper/2407.00634","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.00634"}},"official":{"repos":["bytedance/tarsier"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/stllava-med-self-training-large-language-and","slug":"stllava-med-self-training-large-language-and","title":"STLLaVA-Med: Self-Training Large Language and Vision Assistant for Medical Question-Answering","date":"2024-06-28","arxiv_id":"2406.19973","repositories_listed":1,"syntology":{"n":19,"n_ran":9,"n_constructed":4,"n_ran_checked":5,"n_instrument":4,"n_unverified":10,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"9 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 4 where Syntology's instrument failed) · 10 unverified","sample_list":"/paper/stllava-med-self-training-large-language-and#ran","syntology_url":"https://syntology.ai/paper/2406.19973","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.19973"}},"official":{"repos":["heliossun/stllava-med"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":10,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-continual-learning-in-visual","slug":"enhancing-continual-learning-in-visual","title":"Enhancing Continual Learning in Visual Question Answering with Modality-Aware Feature Distillation","date":"2024-06-27","arxiv_id":"2406.19297","repositories_listed":1,"syntology":null},{"url":"/paper/huatuogpt-vision-towards-injecting-medical","slug":"huatuogpt-vision-towards-injecting-medical","title":"HuatuoGPT-Vision, Towards Injecting Medical Visual Knowledge into Multimodal LLMs at Scale","date":"2024-06-27","arxiv_id":"2406.19280","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/huatuogpt-vision-towards-injecting-medical#ran","syntology_url":"https://syntology.ai/paper/2406.19280","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.19280"}},"official":{"repos":["freedomintelligence/huatuogpt-vision"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vga-vision-gui-assistant-minimizing","slug":"vga-vision-gui-assistant-minimizing","title":"VGA: Vision GUI Assistant -- Minimizing Hallucinations through Image-Centric Fine-Tuning","date":"2024-06-20","arxiv_id":"2406.14056","repositories_listed":1,"syntology":null},{"url":"/paper/biomedical-visual-instruction-tuning-with","slug":"biomedical-visual-instruction-tuning-with","title":"Biomedical Visual Instruction Tuning with Clinician Preference Alignment","date":"2024-06-19","arxiv_id":"2406.13173","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":13,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/biomedical-visual-instruction-tuning-with#ran","syntology_url":"https://syntology.ai/paper/2406.13173","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.13173"}},"official":{"repos":["mao1207/BioMed-VITAL"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/rationale-based-ensemble-of-multiple-qa","slug":"rationale-based-ensemble-of-multiple-qa","title":"Diversify, Rationalize, and Combine: Ensembling Multiple QA Strategies for Zero-shot Knowledge-based VQA","date":"2024-06-18","arxiv_id":"2406.12746","repositories_listed":1,"syntology":null},{"url":"/paper/mmneuron-discovering-neuron-level-domain","slug":"mmneuron-discovering-neuron-level-domain","title":"MMNeuron: Discovering Neuron-Level Domain-Specific Interpretation in Multimodal Large Language Model","date":"2024-06-17","arxiv_id":"2406.11193","repositories_listed":1,"syntology":null},{"url":"/paper/foodieqa-a-multimodal-dataset-for-fine","slug":"foodieqa-a-multimodal-dataset-for-fine","title":"FoodieQA: A Multimodal Dataset for Fine-Grained Understanding of Chinese Food Culture","date":"2024-06-16","arxiv_id":"2406.11030","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/foodieqa-a-multimodal-dataset-for-fine#ran","syntology_url":"https://syntology.ai/paper/2406.11030","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11030"}},"official":{"repos":["lyan62/FoodieQA"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-raw-videos-understanding-edited-videos","slug":"beyond-raw-videos-understanding-edited-videos","title":"Beyond Raw Videos: Understanding Edited Videos with Large Multimodal Model","date":"2024-06-15","arxiv_id":"2406.10484","repositories_listed":1,"syntology":null},{"url":"/paper/vision-language-models-meet-meteorology","slug":"vision-language-models-meet-meteorology","title":"Vision-Language Models Meet Meteorology: Developing Models for Extreme Weather Events Detection with Heatmaps","date":"2024-06-14","arxiv_id":"2406.09838","repositories_listed":1,"syntology":null},{"url":"/paper/what-is-the-visual-cognition-gap-between","slug":"what-is-the-visual-cognition-gap-between","title":"What is the Visual Cognition Gap between Humans and Multimodal LLMs?","date":"2024-06-14","arxiv_id":"2406.10424","repositories_listed":1,"syntology":null},{"url":"/paper/visionllm-v2-an-end-to-end-generalist","slug":"visionllm-v2-an-end-to-end-generalist","title":"VisionLLM v2: An End-to-End Generalist Multimodal Large Language Model for Hundreds of Vision-Language Tasks","date":"2024-06-12","arxiv_id":"2406.08394","repositories_listed":1,"syntology":null},{"url":"/paper/composition-vision-language-understanding-via","slug":"composition-vision-language-understanding-via","title":"Composition Vision-Language Understanding via Segment and Depth Anything Model","date":"2024-06-07","arxiv_id":"2406.18591","repositories_listed":1,"syntology":null},{"url":"/paper/diffusion-refined-vqa-annotations-for-semi","slug":"diffusion-refined-vqa-annotations-for-semi","title":"Diffusion-Refined VQA Annotations for Semi-Supervised Gaze Following","date":"2024-06-04","arxiv_id":"2406.02774","repositories_listed":1,"syntology":null},{"url":"/paper/dragonfly-multi-resolution-zoom-supercharges","slug":"dragonfly-multi-resolution-zoom-supercharges","title":"Dragonfly: Multi-Resolution Zoom-In Encoding Enhances Vision-Language Models","date":"2024-06-03","arxiv_id":"2406.00977","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dragonfly-multi-resolution-zoom-supercharges#ran","syntology_url":"https://syntology.ai/paper/2406.00977","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.00977"}},"official":{"repos":["togethercomputer/dragonfly"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tabpedia-towards-comprehensive-visual-table","slug":"tabpedia-towards-comprehensive-visual-table","title":"TabPedia: Towards Comprehensive Visual Table Understanding with Concept Synergy","date":"2024-06-03","arxiv_id":"2406.01326","repositories_listed":1,"syntology":null},{"url":"/paper/deco-decoupling-token-compression-from","slug":"deco-decoupling-token-compression-from","title":"DeCo: Decoupling Token Compression from Semantic Abstraction in Multimodal Large Language Models","date":"2024-05-31","arxiv_id":"2405.20985","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/deco-decoupling-token-compression-from#ran","syntology_url":"https://syntology.ai/paper/2405.20985","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.20985"}},"official":{"repos":["yaolinli/deco"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/instruction-guided-visual-masking","slug":"instruction-guided-visual-masking","title":"Instruction-Guided Visual Masking","date":"2024-05-30","arxiv_id":"2405.19783","repositories_listed":1,"syntology":{"n":28,"n_ran":20,"n_constructed":7,"n_ran_checked":11,"n_instrument":9,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":2,"phrase":"20 ran (of which 7 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 9 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/instruction-guided-visual-masking#ran","syntology_url":"https://syntology.ai/paper/2405.19783","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19783"}},"official":{"repos":["2toinf/ivm"],"state":"official (archive's flag): 20 ran","n_ran":20,"n_constructed":7,"n_ran_no_instrument_failure":11,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/worse-than-random-an-embarrassingly-simple","slug":"worse-than-random-an-embarrassingly-simple","title":"Worse than Random? An Embarrassingly Simple Probing Evaluation of Large Multimodal Models in Medical VQA","date":"2024-05-30","arxiv_id":"2405.20421","repositories_listed":1,"syntology":null},{"url":"/paper/reverse-image-retrieval-cues-parametric","slug":"reverse-image-retrieval-cues-parametric","title":"Reverse Image Retrieval Cues Parametric Memory in Multimodal LLMs","date":"2024-05-29","arxiv_id":"2405.18740","repositories_listed":1,"syntology":null},{"url":"/paper/lm4lv-a-frozen-large-language-model-for-low","slug":"lm4lv-a-frozen-large-language-model-for-low","title":"LM4LV: A Frozen Large Language Model for Low-level Vision Tasks","date":"2024-05-24","arxiv_id":"2405.15734","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/lm4lv-a-frozen-large-language-model-for-low#ran","syntology_url":"https://syntology.ai/paper/2405.15734","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.15734"}},"official":{"repos":["bytetriper/lm4lv"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/pitvqa-image-grounded-text-embedding-llm-for","slug":"pitvqa-image-grounded-text-embedding-llm-for","title":"PitVQA: Image-grounded Text Embedding LLM for Visual Question Answering in Pituitary Surgery","date":"2024-05-22","arxiv_id":"2405.13949","repositories_listed":1,"syntology":null},{"url":"/paper/dataset-and-benchmark-for-urdu-natural-scenes","slug":"dataset-and-benchmark-for-urdu-natural-scenes","title":"Dataset and Benchmark for Urdu Natural Scenes Text Detection, Recognition and Visual Question Answering","date":"2024-05-21","arxiv_id":"2405.12533","repositories_listed":1,"syntology":null},{"url":"/paper/mtvqa-benchmarking-multilingual-text-centric","slug":"mtvqa-benchmarking-multilingual-text-centric","title":"MTVQA: Benchmarking Multilingual Text-Centric Visual Question Answering","date":"2024-05-20","arxiv_id":"2405.11985","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mtvqa-benchmarking-multilingual-text-centric#ran","syntology_url":"https://syntology.ai/paper/2405.11985","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.11985"}},"official":{"repos":["bytedance/MTVQA"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/federated-document-visual-question-answering","slug":"federated-document-visual-question-answering","title":"Federated Document Visual Question Answering: A Pilot Study","date":"2024-05-10","arxiv_id":"2405.06636","repositories_listed":1,"syntology":null},{"url":"/paper/cumo-scaling-multimodal-llm-with-co-upcycled","slug":"cumo-scaling-multimodal-llm-with-co-upcycled","title":"CuMo: Scaling Multimodal LLM with Co-Upcycled Mixture-of-Experts","date":"2024-05-09","arxiv_id":"2405.05949","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":6,"n_instrument":5,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cumo-scaling-multimodal-llm-with-co-upcycled#ran","syntology_url":"https://syntology.ai/paper/2405.05949","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.05949"}},"official":{"repos":["shi-labs/cumo"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/exploring-the-capabilities-of-large","slug":"exploring-the-capabilities-of-large","title":"Exploring the Capabilities of Large Multimodal Models on Dense Text","date":"2024-05-09","arxiv_id":"2405.06706","repositories_listed":1,"syntology":null},{"url":"/paper/light-vqa-a-video-quality-assessment-model","slug":"light-vqa-a-video-quality-assessment-model","title":"Light-VQA+: A Video Quality Assessment Model for Exposure Correction with Vision-Language Guidance","date":"2024-05-06","arxiv_id":"2405.03333","repositories_listed":1,"syntology":null},{"url":"/paper/omnidrive-a-holistic-llm-agent-framework-for","slug":"omnidrive-a-holistic-llm-agent-framework-for","title":"OmniDrive: A Holistic Vision-Language Dataset for Autonomous Driving with Counterfactual Reasoning","date":"2024-05-02","arxiv_id":"2405.01533","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/omnidrive-a-holistic-llm-agent-framework-for#ran","syntology_url":"https://syntology.ai/paper/2405.01533","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.01533"}},"official":{"repos":["nvlabs/omnidrive"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-page-document-visual-question-answering","slug":"multi-page-document-visual-question-answering","title":"Multi-Page Document Visual Question Answering using Self-Attention Scoring Mechanism","date":"2024-04-29","arxiv_id":"2404.19024","repositories_listed":1,"syntology":null},{"url":"/paper/radgenome-chest-ct-a-grounded-vision-language","slug":"radgenome-chest-ct-a-grounded-vision-language","title":"RadGenome-Chest CT: A Grounded Vision-Language Dataset for Chest CT Analysis","date":"2024-04-25","arxiv_id":"2404.16754","repositories_listed":1,"syntology":null},{"url":"/paper/ais-2024-challenge-on-video-quality","slug":"ais-2024-challenge-on-video-quality","title":"AIS 2024 Challenge on Video Quality Assessment of User-Generated Content: Methods and Results","date":"2024-04-24","arxiv_id":"2404.16205","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-collaboration-strategy-for-llms-in","slug":"adaptive-collaboration-strategy-for-llms-in","title":"MDAgents: An Adaptive Collaboration of LLMs for Medical Decision-Making","date":"2024-04-22","arxiv_id":"2404.15155","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adaptive-collaboration-strategy-for-llms-in#ran","syntology_url":"https://syntology.ai/paper/2404.15155","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.15155"}},"official":{"repos":["mitmedialab/mdagents"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/boter-bootstrapping-knowledge-selection-and","slug":"boter-bootstrapping-knowledge-selection-and","title":"Self-Bootstrapped Visual-Language Model for Knowledge Selection and Question Answering","date":"2024-04-22","arxiv_id":"2404.13947","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/boter-bootstrapping-knowledge-selection-and#ran","syntology_url":"https://syntology.ai/paper/2404.13947","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.13947"}},"official":{"repos":["haodongze/self-ksel-qans"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/lapa-latent-prompt-assist-model-for-medical","slug":"lapa-latent-prompt-assist-model-for-medical","title":"LaPA: Latent Prompt Assist Model For Medical Visual Question Answering","date":"2024-04-19","arxiv_id":"2404.13039","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":7,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/lapa-latent-prompt-assist-model-for-medical#ran","syntology_url":"https://syntology.ai/paper/2404.13039","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.13039"}},"official":{"repos":["garygutc/lapa_model"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/ntire-2024-challenge-on-short-form-ugc-video","slug":"ntire-2024-challenge-on-short-form-ugc-video","title":"NTIRE 2024 Challenge on Short-form UGC Video Quality Assessment: Methods and Results","date":"2024-04-17","arxiv_id":"2404.11313","repositories_listed":1,"syntology":null},{"url":"/paper/textcot-zoom-in-for-enhanced-multimodal-text","slug":"textcot-zoom-in-for-enhanced-multimodal-text","title":"TextCoT: Zoom In for Enhanced Multimodal Text-Rich Image Understanding","date":"2024-04-15","arxiv_id":"2404.09797","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/textcot-zoom-in-for-enhanced-multimodal-text#ran","syntology_url":"https://syntology.ai/paper/2404.09797","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.09797"}},"official":{"repos":["bzluan/textcot"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-visual-question-answering-through","slug":"enhancing-visual-question-answering-through","title":"Enhancing Visual Question Answering through Question-Driven Image Captions as Prompts","date":"2024-04-12","arxiv_id":"2404.08589","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/enhancing-visual-question-answering-through#ran","syntology_url":"https://syntology.ai/paper/2404.08589","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.08589"}},"official":{"repos":["ovguyo/captions-in-vqa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-localize-objects-improves-spatial","slug":"learning-to-localize-objects-improves-spatial","title":"Learning to Localize Objects Improves Spatial Reasoning in Visual-LLMs","date":"2024-04-11","arxiv_id":"2404.07449","repositories_listed":1,"syntology":null},{"url":"/paper/ma-lmm-memory-augmented-large-multimodal","slug":"ma-lmm-memory-augmented-large-multimodal","title":"MA-LMM: Memory-Augmented Large Multimodal Model for Long-Term Video Understanding","date":"2024-04-08","arxiv_id":"2404.05726","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/ma-lmm-memory-augmented-large-multimodal#ran","syntology_url":"https://syntology.ai/paper/2404.05726","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.05726"}},"official":{"repos":["boheumd/MA-LMM"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/joint-visual-and-text-prompting-for-improved","slug":"joint-visual-and-text-prompting-for-improved","title":"Joint Visual and Text Prompting for Improved Object-Centric Perception with Multimodal Large Language Models","date":"2024-04-06","arxiv_id":"2404.04514","repositories_listed":1,"syntology":null},{"url":"/paper/unsolvable-problem-detection-evaluating","slug":"unsolvable-problem-detection-evaluating","title":"Unsolvable Problem Detection: Evaluating Trustworthiness of Vision Language Models","date":"2024-03-29","arxiv_id":"2403.20331","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/unsolvable-problem-detection-evaluating#ran","syntology_url":"https://syntology.ai/paper/2403.20331","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.20331"}},"official":{"repos":["atsumiyai/upd"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/jdocqa-japanese-document-question-answering","slug":"jdocqa-japanese-document-question-answering","title":"JDocQA: Japanese Document Question Answering Dataset for Generative Language Models","date":"2024-03-28","arxiv_id":"2403.19454","repositories_listed":1,"syntology":null},{"url":"/paper/quantifying-and-mitigating-unimodal-biases-in","slug":"quantifying-and-mitigating-unimodal-biases-in","title":"Quantifying and Mitigating Unimodal Biases in Multimodal Large Language Models: A Causal Perspective","date":"2024-03-27","arxiv_id":"2403.18346","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/quantifying-and-mitigating-unimodal-biases-in#ran","syntology_url":"https://syntology.ai/paper/2403.18346","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.18346"}},"official":{"repos":["opencausalab/more"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/intrinsic-subgraph-generation-for","slug":"intrinsic-subgraph-generation-for","title":"Intrinsic Subgraph Generation for Interpretable Graph based Visual Question Answering","date":"2024-03-26","arxiv_id":"2403.17647","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/intrinsic-subgraph-generation-for#ran","syntology_url":"https://syntology.ai/paper/2403.17647","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.17647"}},"official":{"repos":["digitalphonetics/intrinsic-subgraph-generation-for-vqa"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/visual-cot-unleashing-chain-of-thought","slug":"visual-cot-unleashing-chain-of-thought","title":"Visual CoT: Advancing Multi-Modal Language Models with a Comprehensive Dataset and Benchmark for Chain-of-Thought Reasoning","date":"2024-03-25","arxiv_id":"2403.16999","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":4,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 2 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/visual-cot-unleashing-chain-of-thought#ran","syntology_url":"https://syntology.ai/paper/2403.16999","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.16999"}},"official":{"repos":["deepcs233/visual-cot"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/illusionvqa-a-challenging-optical-illusion","slug":"illusionvqa-a-challenging-optical-illusion","title":"IllusionVQA: A Challenging Optical Illusion Dataset for Vision Language Models","date":"2024-03-23","arxiv_id":"2403.15952","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/illusionvqa-a-challenging-optical-illusion#ran","syntology_url":"https://syntology.ai/paper/2403.15952","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.15952"}},"official":{"repos":["csebuetnlp/illusionvqa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/medpromptx-grounded-multimodal-prompting-for","slug":"medpromptx-grounded-multimodal-prompting-for","title":"MedPromptX: Grounded Multimodal Prompting for Chest X-ray Diagnosis","date":"2024-03-22","arxiv_id":"2403.15585","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-vqa-exploring-multi-agent","slug":"multi-agent-vqa-exploring-multi-agent","title":"Multi-Agent VQA: Exploring Multi-Agent Foundation Models in Zero-Shot Visual Question Answering","date":"2024-03-21","arxiv_id":"2403.14783","repositories_listed":1,"syntology":null},{"url":"/paper/vid-tldr-training-free-token-merging-for","slug":"vid-tldr-training-free-token-merging-for","title":"vid-TLDR: Training Free Token merging for Light-weight Video Transformer","date":"2024-03-20","arxiv_id":"2403.13347","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vid-tldr-training-free-token-merging-for#ran","syntology_url":"https://syntology.ai/paper/2403.13347","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.13347"}},"official":{"repos":["mlvlab/vid-tldr"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/hydra-a-hyper-agent-for-dynamic-compositional","slug":"hydra-a-hyper-agent-for-dynamic-compositional","title":"HYDRA: A Hyper Agent for Dynamic Compositional Visual Reasoning","date":"2024-03-19","arxiv_id":"2403.12884","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/hydra-a-hyper-agent-for-dynamic-compositional#ran","syntology_url":"https://syntology.ai/paper/2403.12884","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.12884"}},"official":{"repos":["ControlNet/HYDRA"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/vl-icl-bench-the-devil-in-the-details-of","slug":"vl-icl-bench-the-devil-in-the-details-of","title":"VL-ICL Bench: The Devil in the Details of Multimodal In-Context Learning","date":"2024-03-19","arxiv_id":"2403.13164","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/vl-icl-bench-the-devil-in-the-details-of#ran","syntology_url":"https://syntology.ai/paper/2403.13164","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.13164"}},"official":{"repos":["ys-zong/vl-icl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/phd-a-prompted-visual-hallucination","slug":"phd-a-prompted-visual-hallucination","title":"PhD: A ChatGPT-Prompted Visual hallucination Evaluation Dataset","date":"2024-03-17","arxiv_id":"2403.11116","repositories_listed":1,"syntology":null},{"url":"/paper/adversarial-training-with-ocr-modality","slug":"adversarial-training-with-ocr-modality","title":"Adversarial Training with OCR Modality Perturbation for Scene-Text Visual Question Answering","date":"2024-03-14","arxiv_id":"2403.09288","repositories_listed":1,"syntology":null},{"url":"/paper/lumen-unleashing-versatile-vision-centric","slug":"lumen-unleashing-versatile-vision-centric","title":"Lumen: Unleashing Versatile Vision-Centric Capabilities of Large Multimodal Models","date":"2024-03-12","arxiv_id":"2403.07304","repositories_listed":1,"syntology":null},{"url":"/paper/multi-modal-auto-regressive-modeling-via","slug":"multi-modal-auto-regressive-modeling-via","title":"Multi-modal Auto-regressive Modeling via Visual Words","date":"2024-03-12","arxiv_id":"2403.07720","repositories_listed":1,"syntology":null},{"url":"/paper/textmonkey-an-ocr-free-large-multimodal-model","slug":"textmonkey-an-ocr-free-large-multimodal-model","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","date":"2024-03-07","arxiv_id":"2403.04473","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/textmonkey-an-ocr-free-large-multimodal-model#ran","syntology_url":"https://syntology.ai/paper/2403.04473","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.04473"}},"official":{"repos":["yuliang-liu/monkey"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vision-language-models-for-medical-report","slug":"vision-language-models-for-medical-report","title":"Vision-Language Models for Medical Report Generation and Visual Question Answering: A Review","date":"2024-03-04","arxiv_id":"2403.02469","repositories_listed":1,"syntology":null},{"url":"/paper/llm-assisted-multi-teacher-continual-learning","slug":"llm-assisted-multi-teacher-continual-learning","title":"LLM-Assisted Multi-Teacher Continual Learning for Visual Question Answering in Robotic Surgery","date":"2024-02-26","arxiv_id":"2402.16664","repositories_listed":1,"syntology":null},{"url":"/paper/bridging-the-gap-between-2d-and-3d-visual","slug":"bridging-the-gap-between-2d-and-3d-visual","title":"Bridging the Gap between 2D and 3D Visual Question Answering: A Fusion Approach for 3D VQA","date":"2024-02-24","arxiv_id":"2402.15933","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":8,"n_instrument":4,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":13,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bridging-the-gap-between-2d-and-3d-visual#ran","syntology_url":"https://syntology.ai/paper/2402.15933","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.15933"}},"official":{"repos":["matthewdm0816/bridgeqa"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/commvqa-situating-visual-question-answering","slug":"commvqa-situating-visual-question-answering","title":"CommVQA: Situating Visual Question Answering in Communicative Contexts","date":"2024-02-22","arxiv_id":"2402.15002","repositories_listed":1,"syntology":null},{"url":"/paper/uncertainty-aware-evaluation-for-vision","slug":"uncertainty-aware-evaluation-for-vision","title":"Uncertainty-Aware Evaluation for Vision-Language Models","date":"2024-02-22","arxiv_id":"2402.14418","repositories_listed":1,"syntology":{"n":17,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":9,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/uncertainty-aware-evaluation-for-vision#ran","syntology_url":"https://syntology.ai/paper/2402.14418","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14418"}},"official":{"repos":["ensec-ai/vlm-uncertainty-bench"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/cognitive-visual-language-mapper-advancing","slug":"cognitive-visual-language-mapper-advancing","title":"Cognitive Visual-Language Mapper: Advancing Multimodal Comprehension with Enhanced Visual Knowledge Alignment","date":"2024-02-21","arxiv_id":"2402.13561","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/cognitive-visual-language-mapper-advancing#ran","syntology_url":"https://syntology.ai/paper/2402.13561","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.13561"}},"official":{"repos":["hitsz-tmg/cognitive-visual-language-mapper"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/collavo-crayon-large-language-and-vision","slug":"collavo-crayon-large-language-and-vision","title":"CoLLaVO: Crayon Large Language and Vision mOdel","date":"2024-02-17","arxiv_id":"2402.11248","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/collavo-crayon-large-language-and-vision#ran","syntology_url":"https://syntology.ai/paper/2402.11248","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11248"}},"official":{"repos":["ByungKwanLee/CoLLaVO"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ii-mmr-identifying-and-improving-multi-modal","slug":"ii-mmr-identifying-and-improving-multi-modal","title":"II-MMR: Identifying and Improving Multi-modal Multi-hop Reasoning in Visual Question Answering","date":"2024-02-16","arxiv_id":"2402.11058","repositories_listed":1,"syntology":null},{"url":"/paper/omnimedvqa-a-new-large-scale-comprehensive","slug":"omnimedvqa-a-new-large-scale-comprehensive","title":"OmniMedVQA: A New Large-Scale Comprehensive Evaluation Benchmark for Medical LVLM","date":"2024-02-14","arxiv_id":"2402.09181","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/omnimedvqa-a-new-large-scale-comprehensive#ran","syntology_url":"https://syntology.ai/paper/2402.09181","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.09181"}},"official":{"repos":["opengvlab/multi-modality-arena"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/pretraining-vision-language-model-for","slug":"pretraining-vision-language-model-for","title":"Pretraining Vision-Language Model for Difference Visual Question Answering in Longitudinal Chest X-rays","date":"2024-02-14","arxiv_id":"2402.08966","repositories_listed":1,"syntology":null},{"url":"/paper/preflmr-scaling-up-fine-grained-late","slug":"preflmr-scaling-up-fine-grained-late","title":"PreFLMR: Scaling Up Fine-Grained Late-Interaction Multi-modal Retrievers","date":"2024-02-13","arxiv_id":"2402.08327","repositories_listed":1,"syntology":null},{"url":"/paper/kvq-kaleidoscope-video-quality-assessment-for","slug":"kvq-kaleidoscope-video-quality-assessment-for","title":"KVQ: Kwai Video Quality Assessment for Short-form Videos","date":"2024-02-11","arxiv_id":"2402.07220","repositories_listed":1,"syntology":{"n":28,"n_ran":22,"n_constructed":10,"n_ran_checked":15,"n_instrument":7,"n_unverified":6,"n_honours":1,"n_violates":0,"n_no_contract":14,"n_pointer_only":28,"phrase":"22 ran (of which 10 constructed an object rather than computing a result; 15 with no instrument failure: 1 honoured, 0 violated, 14 with no contract checked; 7 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/kvq-kaleidoscope-video-quality-assessment-for#ran","syntology_url":"https://syntology.ai/paper/2402.07220","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.07220"}},"official":null}},{"url":"/paper/open-ended-vqa-benchmarking-of-vision","slug":"open-ended-vqa-benchmarking-of-vision","title":"Open-ended VQA benchmarking of Vision-Language models by exploiting Classification datasets and their semantic hierarchy","date":"2024-02-11","arxiv_id":"2402.07270","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/open-ended-vqa-benchmarking-of-vision#ran","syntology_url":"https://syntology.ai/paper/2402.07270","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.07270"}},"official":{"repos":["lmb-freiburg/ovqa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/gemini-goes-to-med-school-exploring-the","slug":"gemini-goes-to-med-school-exploring-the","title":"Gemini Goes to Med School: Exploring the Capabilities of Multimodal Large Language Models on Medical Challenge Problems & Hallucinations","date":"2024-02-10","arxiv_id":"2402.07023","repositories_listed":1,"syntology":null},{"url":"/paper/convincing-rationales-for-visual-question","slug":"convincing-rationales-for-visual-question","title":"Convincing Rationales for Visual Question Answering Reasoning","date":"2024-02-06","arxiv_id":"2402.03896","repositories_listed":1,"syntology":null},{"url":"/paper/text-guided-image-clustering","slug":"text-guided-image-clustering","title":"Text-Guided Image Clustering","date":"2024-02-05","arxiv_id":"2402.02996","repositories_listed":1,"syntology":null},{"url":"/paper/video-lavit-unified-video-language-pre","slug":"video-lavit-unified-video-language-pre","title":"Video-LaVIT: Unified Video-Language Pre-training with Decoupled Visual-Motional Tokenization","date":"2024-02-05","arxiv_id":"2402.03161","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/video-lavit-unified-video-language-pre#ran","syntology_url":"https://syntology.ai/paper/2402.03161","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.03161"}},"official":null}},{"url":"/paper/gerea-question-aware-prompt-captions-for","slug":"gerea-question-aware-prompt-captions-for","title":"GeReA: Question-Aware Prompt Captions for Knowledge-based Visual Question Answering","date":"2024-02-04","arxiv_id":"2402.02503","repositories_listed":1,"syntology":{"n":18,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":18,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/gerea-question-aware-prompt-captions-for#ran","syntology_url":"https://syntology.ai/paper/2402.02503","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.02503"}},"official":{"repos":["upper9527/gerea"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/knowledge-generation-for-zero-shot-knowledge","slug":"knowledge-generation-for-zero-shot-knowledge","title":"Knowledge Generation for Zero-shot Knowledge-based VQA","date":"2024-02-04","arxiv_id":"2402.02541","repositories_listed":1,"syntology":null},{"url":"/paper/common-sense-reasoning-for-deep-fake","slug":"common-sense-reasoning-for-deep-fake","title":"Common Sense Reasoning for Deepfake Detection","date":"2024-01-31","arxiv_id":"2402.00126","repositories_listed":1,"syntology":null},{"url":"/paper/proximity-qa-unleashing-the-power-of-multi","slug":"proximity-qa-unleashing-the-power-of-multi","title":"Proximity QA: Unleashing the Power of Multi-Modal Large Language Models for Spatial Proximity Analysis","date":"2024-01-31","arxiv_id":"2401.17862","repositories_listed":1,"syntology":null},{"url":"/paper/q-a-prompts-discovering-rich-visual-clues","slug":"q-a-prompts-discovering-rich-visual-clues","title":"Q&A Prompts: Discovering Rich Visual Clues through Mining Question-Answer Prompts for VQA requiring Diverse World Knowledge","date":"2024-01-19","arxiv_id":"2401.10712","repositories_listed":1,"syntology":null},{"url":"/paper/question-answer-cross-language-image-matching","slug":"question-answer-cross-language-image-matching","title":"Question-Answer Cross Language Image Matching for Weakly Supervised Semantic Segmentation","date":"2024-01-18","arxiv_id":"2401.09883","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/question-answer-cross-language-image-matching#ran","syntology_url":"https://syntology.ai/paper/2401.09883","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.09883"}},"official":{"repos":["cvi-szu/qa-clims"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/veagle-advancements-in-multimodal","slug":"veagle-advancements-in-multimodal","title":"Veagle: Advancements in Multimodal Representation Learning","date":"2024-01-18","arxiv_id":"2403.08773","repositories_listed":1,"syntology":null},{"url":"/paper/uncovering-the-full-potential-of-visual","slug":"uncovering-the-full-potential-of-visual","title":"Uncovering the Full Potential of Visual Grounding Methods in VQA","date":"2024-01-15","arxiv_id":"2401.07803","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/uncovering-the-full-potential-of-visual#ran","syntology_url":"https://syntology.ai/paper/2401.07803","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.07803"}},"official":{"repos":["dreichcsl/truevg"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/generalizing-visual-question-answering-from","slug":"generalizing-visual-question-answering-from","title":"Generalizing Visual Question Answering from Synthetic to Human-Written Questions via a Chain of QA with a Large Language Model","date":"2024-01-12","arxiv_id":"2401.06400","repositories_listed":1,"syntology":null},{"url":"/paper/cross-modal-retrieval-for-knowledge-based","slug":"cross-modal-retrieval-for-knowledge-based","title":"Cross-modal Retrieval for Knowledge-based Visual Question Answering","date":"2024-01-11","arxiv_id":"2401.05736","repositories_listed":1,"syntology":{"n":11,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":11,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/cross-modal-retrieval-for-knowledge-based#ran","syntology_url":"https://syntology.ai/paper/2401.05736","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.05736"}},"official":{"repos":["paullerner/viquae"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/hallucination-benchmark-in-medical-visual","slug":"hallucination-benchmark-in-medical-visual","title":"Hallucination Benchmark in Medical Visual Question Answering","date":"2024-01-11","arxiv_id":"2401.05827","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hallucination-benchmark-in-medical-visual#ran","syntology_url":"https://syntology.ai/paper/2401.05827","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.05827"}},"official":{"repos":["knowlab/halt-medvqa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/miss-a-generative-pretraining-and-finetuning","slug":"miss-a-generative-pretraining-and-finetuning","title":"MISS: A Generative Pretraining and Finetuning Approach for Med-VQA","date":"2024-01-10","arxiv_id":"2401.05163","repositories_listed":1,"syntology":null},{"url":"/paper/3dmit-3d-multi-modal-instruction-tuning-for","slug":"3dmit-3d-multi-modal-instruction-tuning-for","title":"3DMIT: 3D Multi-modal Instruction Tuning for Scene Understanding","date":"2024-01-06","arxiv_id":"2401.03201","repositories_listed":1,"syntology":{"n":10,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":10,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/3dmit-3d-multi-modal-instruction-tuning-for#ran","syntology_url":"https://syntology.ai/paper/2401.03201","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.03201"}},"official":{"repos":["staymylove/3DMIT"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/pefomed-parameter-efficient-fine-tuning-on","slug":"pefomed-parameter-efficient-fine-tuning-on","title":"PeFoMed: Parameter Efficient Fine-tuning of Multimodal Large Language Models for Medical Imaging","date":"2024-01-05","arxiv_id":"2401.02797","repositories_listed":1,"syntology":null}],"record_sha256":"15c2a0ad7b72df78967481c549214dc42fa40ddf4c1003df2799255af12c2ef3","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}