{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/visual-question-answering-1/papers/6","list_of":"/task/visual-question-answering-1","task":"Visual Question Answering","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":6,"pages_in_order":22,"rows_per_page":100,"rows":[501,600],"of":2177,"counts":{"archive_papers_tagged":2177,"with_a_code_link":1042,"where_syntology_ran_a_sample":378,"not_listed_spam_title":0,"listed":2177,"listed_where_code_ran":378,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":308,"every_run_a_failure_of_syntologys_instrument":70,"listed_with_a_run_with_no_instrument_failure":308,"listed_every_run_a_failure_of_syntologys_instrument":70,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/visual-question-answering-1","prev":"/task/visual-question-answering-1/papers/5","next":"/task/visual-question-answering-1/papers/7","papers":[{"url":"/paper/self-supervised-visual-preference-alignment","slug":"self-supervised-visual-preference-alignment","title":"Self-Supervised Visual Preference Alignment","date":"2024-04-16","arxiv_id":"2404.10501","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":6,"n_instrument":4,"n_unverified":1,"n_honours":1,"n_violates":1,"n_no_contract":4,"n_pointer_only":11,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 1 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/self-supervised-visual-preference-alignment#ran","syntology_url":"https://syntology.ai/paper/2404.10501","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.10501"}},"official":{"repos":["Kevinz-code/SeVa"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/bridging-vision-and-language-spaces-with","slug":"bridging-vision-and-language-spaces-with","title":"Bridging Vision and Language Spaces with Assignment Prediction","date":"2024-04-15","arxiv_id":"2404.09632","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bridging-vision-and-language-spaces-with#ran","syntology_url":"https://syntology.ai/paper/2404.09632","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.09632"}},"official":{"repos":["park-jungin/vlap"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-visual-question-answering-through","slug":"enhancing-visual-question-answering-through","title":"Enhancing Visual Question Answering through Question-Driven Image Captions as Prompts","date":"2024-04-12","arxiv_id":"2404.08589","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/enhancing-visual-question-answering-through#ran","syntology_url":"https://syntology.ai/paper/2404.08589","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.08589"}},"official":{"repos":["ovguyo/captions-in-vqa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ferret-v2-an-improved-baseline-for-referring","slug":"ferret-v2-an-improved-baseline-for-referring","title":"Ferret-v2: An Improved Baseline for Referring and Grounding with Large Language Models","date":"2024-04-11","arxiv_id":"2404.07973","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ferret-v2-an-improved-baseline-for-referring#ran","syntology_url":"https://syntology.ai/paper/2404.07973","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.07973"}},"official":null}},{"url":"/paper/learning-to-localize-objects-improves-spatial","slug":"learning-to-localize-objects-improves-spatial","title":"Learning to Localize Objects Improves Spatial Reasoning in Visual-LLMs","date":"2024-04-11","arxiv_id":"2404.07449","repositories_listed":1,"syntology":null},{"url":"/paper/multi-image-visual-question-answering-for","slug":"multi-image-visual-question-answering-for","title":"Language Models Meet Anomaly Detection for Better Interpretability and Generalizability","date":"2024-04-11","arxiv_id":"2404.07622","repositories_listed":1,"syntology":null},{"url":"/paper/joint-visual-and-text-prompting-for-improved","slug":"joint-visual-and-text-prompting-for-improved","title":"Joint Visual and Text Prompting for Improved Object-Centric Perception with Multimodal Large Language Models","date":"2024-04-06","arxiv_id":"2404.04514","repositories_listed":1,"syntology":null},{"url":"/paper/soft-prompting-with-graph-of-thought-for","slug":"soft-prompting-with-graph-of-thought-for","title":"Soft-Prompting with Graph-of-Thought for Multi-modal Representation Learning","date":"2024-04-06","arxiv_id":"2404.04538","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/soft-prompting-with-graph-of-thought-for#ran","syntology_url":"https://syntology.ai/paper/2404.04538","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.04538"}},"official":{"repos":["shishicode/agot"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/causalchaos-dataset-for-comprehensive-causal","slug":"causalchaos-dataset-for-comprehensive-causal","title":"CausalChaos! Dataset for Comprehensive Causal Action Question Answering Over Longer Causal Chains Grounded in Dynamic Visual Scenes","date":"2024-04-01","arxiv_id":"2404.01299","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/causalchaos-dataset-for-comprehensive-causal#ran","syntology_url":"https://syntology.ai/paper/2404.01299","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.01299"}},"official":{"repos":["lunaproject22/causalchaos"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-by-correction-efficient-tuning-task","slug":"learning-by-correction-efficient-tuning-task","title":"Learning by Correction: Efficient Tuning Task for Zero-Shot Generative Vision-Language Reasoning","date":"2024-04-01","arxiv_id":"2404.00909","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":1,"n_ran_checked":4,"n_instrument":4,"n_unverified":2,"n_honours":1,"n_violates":2,"n_no_contract":1,"n_pointer_only":0,"phrase":"8 ran (of which 1 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 2 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/learning-by-correction-efficient-tuning-task#ran","syntology_url":"https://syntology.ai/paper/2404.00909","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.00909"}},"official":{"repos":["shtuplus/iccc_cvpr2024"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":1,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/unsolvable-problem-detection-evaluating","slug":"unsolvable-problem-detection-evaluating","title":"Unsolvable Problem Detection: Evaluating Trustworthiness of Vision Language Models","date":"2024-03-29","arxiv_id":"2403.20331","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/unsolvable-problem-detection-evaluating#ran","syntology_url":"https://syntology.ai/paper/2403.20331","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.20331"}},"official":{"repos":["atsumiyai/upd"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/jdocqa-japanese-document-question-answering","slug":"jdocqa-japanese-document-question-answering","title":"JDocQA: Japanese Document Question Answering Dataset for Generative Language Models","date":"2024-03-28","arxiv_id":"2403.19454","repositories_listed":1,"syntology":null},{"url":"/paper/multi-frame-lightweight-efficient-vision","slug":"multi-frame-lightweight-efficient-vision","title":"Multi-Frame, Lightweight & Efficient Vision-Language Models for Question Answering in Autonomous Driving","date":"2024-03-28","arxiv_id":"2403.19838","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-embeddings-the-promise-of-visual-table","slug":"beyond-embeddings-the-promise-of-visual-table","title":"Beyond Embeddings: The Promise of Visual Table in Visual Reasoning","date":"2024-03-27","arxiv_id":"2403.18252","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":2,"n_no_contract":3,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 2 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/beyond-embeddings-the-promise-of-visual-table#ran","syntology_url":"https://syntology.ai/paper/2403.18252","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.18252"}},"official":{"repos":["lavi-lab/visual-table"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/quantifying-and-mitigating-unimodal-biases-in","slug":"quantifying-and-mitigating-unimodal-biases-in","title":"Quantifying and Mitigating Unimodal Biases in Multimodal Large Language Models: A Causal Perspective","date":"2024-03-27","arxiv_id":"2403.18346","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/quantifying-and-mitigating-unimodal-biases-in#ran","syntology_url":"https://syntology.ai/paper/2403.18346","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.18346"}},"official":{"repos":["opencausalab/more"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/intrinsic-subgraph-generation-for","slug":"intrinsic-subgraph-generation-for","title":"Intrinsic Subgraph Generation for Interpretable Graph based Visual Question Answering","date":"2024-03-26","arxiv_id":"2403.17647","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/intrinsic-subgraph-generation-for#ran","syntology_url":"https://syntology.ai/paper/2403.17647","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.17647"}},"official":{"repos":["digitalphonetics/intrinsic-subgraph-generation-for-vqa"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/illusionvqa-a-challenging-optical-illusion","slug":"illusionvqa-a-challenging-optical-illusion","title":"IllusionVQA: A Challenging Optical Illusion Dataset for Vision Language Models","date":"2024-03-23","arxiv_id":"2403.15952","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/illusionvqa-a-challenging-optical-illusion#ran","syntology_url":"https://syntology.ai/paper/2403.15952","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.15952"}},"official":{"repos":["csebuetnlp/illusionvqa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/llava-prumerge-adaptive-token-reduction-for","slug":"llava-prumerge-adaptive-token-reduction-for","title":"LLaVA-PruMerge: Adaptive Token Reduction for Efficient Large Multimodal Models","date":"2024-03-22","arxiv_id":"2403.15388","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/llava-prumerge-adaptive-token-reduction-for#ran","syntology_url":"https://syntology.ai/paper/2403.15388","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.15388"}},"official":null}},{"url":"/paper/medpromptx-grounded-multimodal-prompting-for","slug":"medpromptx-grounded-multimodal-prompting-for","title":"MedPromptX: Grounded Multimodal Prompting for Chest X-ray Diagnosis","date":"2024-03-22","arxiv_id":"2403.15585","repositories_listed":1,"syntology":null},{"url":"/paper/language-repository-for-long-video","slug":"language-repository-for-long-video","title":"Language Repository for Long Video Understanding","date":"2024-03-21","arxiv_id":"2403.14622","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/language-repository-for-long-video#ran","syntology_url":"https://syntology.ai/paper/2403.14622","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.14622"}},"official":{"repos":["kkahatapitiya/langrepo"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-agent-vqa-exploring-multi-agent","slug":"multi-agent-vqa-exploring-multi-agent","title":"Multi-Agent VQA: Exploring Multi-Agent Foundation Models in Zero-Shot Visual Question Answering","date":"2024-03-21","arxiv_id":"2403.14783","repositories_listed":1,"syntology":null},{"url":"/paper/hyperllava-dynamic-visual-and-language-expert","slug":"hyperllava-dynamic-visual-and-language-expert","title":"HyperLLaVA: Dynamic Visual and Language Expert Tuning for Multimodal Large Language Models","date":"2024-03-20","arxiv_id":"2403.13447","repositories_listed":1,"syntology":null},{"url":"/paper/chain-of-spot-interactive-reasoning-improves","slug":"chain-of-spot-interactive-reasoning-improves","title":"Chain-of-Spot: Interactive Reasoning Improves Large Vision-Language Models","date":"2024-03-19","arxiv_id":"2403.12966","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/chain-of-spot-interactive-reasoning-improves#ran","syntology_url":"https://syntology.ai/paper/2403.12966","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.12966"}},"official":{"repos":["dongyh20/chain-of-spot"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vl-icl-bench-the-devil-in-the-details-of","slug":"vl-icl-bench-the-devil-in-the-details-of","title":"VL-ICL Bench: The Devil in the Details of Multimodal In-Context Learning","date":"2024-03-19","arxiv_id":"2403.13164","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/vl-icl-bench-the-devil-in-the-details-of#ran","syntology_url":"https://syntology.ai/paper/2403.13164","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.13164"}},"official":{"repos":["ys-zong/vl-icl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/adversarial-training-with-ocr-modality","slug":"adversarial-training-with-ocr-modality","title":"Adversarial Training with OCR Modality Perturbation for Scene-Text Visual Question Answering","date":"2024-03-14","arxiv_id":"2403.09288","repositories_listed":1,"syntology":null},{"url":"/paper/are-vision-language-models-texture-or-shape","slug":"are-vision-language-models-texture-or-shape","title":"Can We Talk Models Into Seeing the World Differently?","date":"2024-03-14","arxiv_id":"2403.09193","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/are-vision-language-models-texture-or-shape#ran","syntology_url":"https://syntology.ai/paper/2403.09193","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.09193"}},"official":{"repos":["paulgavrikov/vlm_shapebias"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/beyond-text-frozen-large-language-models-in","slug":"beyond-text-frozen-large-language-models-in","title":"Beyond Text: Frozen Large Language Models in Visual Signal Comprehension","date":"2024-03-12","arxiv_id":"2403.07874","repositories_listed":1,"syntology":{"n":26,"n_ran":17,"n_constructed":0,"n_ran_checked":6,"n_instrument":11,"n_unverified":9,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":26,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 11 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/beyond-text-frozen-large-language-models-in#ran","syntology_url":"https://syntology.ai/paper/2403.07874","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.07874"}},"official":{"repos":["zh460045050/v2l-tokenizer"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/moai-mixture-of-all-intelligence-for-large","slug":"moai-mixture-of-all-intelligence-for-large","title":"MoAI: Mixture of All Intelligence for Large Language and Vision Models","date":"2024-03-12","arxiv_id":"2403.07508","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/moai-mixture-of-all-intelligence-for-large#ran","syntology_url":"https://syntology.ai/paper/2403.07508","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.07508"}},"official":{"repos":["ByungKwanLee/MoAI"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-modal-auto-regressive-modeling-via","slug":"multi-modal-auto-regressive-modeling-via","title":"Multi-modal Auto-regressive Modeling via Visual Words","date":"2024-03-12","arxiv_id":"2403.07720","repositories_listed":1,"syntology":null},{"url":"/paper/answering-diverse-questions-via-text-attached","slug":"answering-diverse-questions-via-text-attached","title":"Answering Diverse Questions via Text Attached with Key Audio-Visual Clues","date":"2024-03-11","arxiv_id":"2403.06679","repositories_listed":1,"syntology":null},{"url":"/paper/a-comprehensive-overhaul-of-multimodal","slug":"a-comprehensive-overhaul-of-multimodal","title":"Mipha: A Comprehensive Overhaul of Multimodal Assistant with Small Language Models","date":"2024-03-10","arxiv_id":"2403.06199","repositories_listed":1,"syntology":null},{"url":"/paper/deepseek-vl-towards-real-world-vision","slug":"deepseek-vl-towards-real-world-vision","title":"DeepSeek-VL: Towards Real-World Vision-Language Understanding","date":"2024-03-08","arxiv_id":"2403.05525","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":7,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deepseek-vl-towards-real-world-vision#ran","syntology_url":"https://syntology.ai/paper/2403.05525","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.05525"}},"official":{"repos":["deepseek-ai/deepseek-vl"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/gemini-1-5-unlocking-multimodal-understanding","slug":"gemini-1-5-unlocking-multimodal-understanding","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","date":"2024-03-08","arxiv_id":"2403.05530","repositories_listed":1,"syntology":null},{"url":"/paper/cat-enhancing-multimodal-large-language-model","slug":"cat-enhancing-multimodal-large-language-model","title":"CAT: Enhancing Multimodal Large Language Model to Answer Questions in Dynamic Audio-Visual Scenarios","date":"2024-03-07","arxiv_id":"2403.04640","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/cat-enhancing-multimodal-large-language-model#ran","syntology_url":"https://syntology.ai/paper/2403.04640","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.04640"}},"official":{"repos":["rikeilong/bay-cat"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/feast-your-eyes-mixture-of-resolution","slug":"feast-your-eyes-mixture-of-resolution","title":"Feast Your Eyes: Mixture-of-Resolution Adaptation for Multimodal Large Language Models","date":"2024-03-05","arxiv_id":"2403.03003","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/feast-your-eyes-mixture-of-resolution#ran","syntology_url":"https://syntology.ai/paper/2403.03003","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.03003"}},"official":{"repos":["luogen1996/llava-hr"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vision-language-models-for-medical-report","slug":"vision-language-models-for-medical-report","title":"Vision-Language Models for Medical Report Generation and Visual Question Answering: A Review","date":"2024-03-04","arxiv_id":"2403.02469","repositories_listed":1,"syntology":null},{"url":"/paper/the-all-seeing-project-v2-towards-general","slug":"the-all-seeing-project-v2-towards-general","title":"The All-Seeing Project V2: Towards General Relation Comprehension of the Open World","date":"2024-02-29","arxiv_id":"2402.19474","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":2,"n_ran_checked":5,"n_instrument":2,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":3,"n_pointer_only":8,"phrase":"7 ran (of which 2 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-all-seeing-project-v2-towards-general#ran","syntology_url":"https://syntology.ai/paper/2402.19474","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.19474"}},"official":{"repos":["opengvlab/all-seeing"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":2,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/llm-assisted-multi-teacher-continual-learning","slug":"llm-assisted-multi-teacher-continual-learning","title":"LLM-Assisted Multi-Teacher Continual Learning for Visual Question Answering in Robotic Surgery","date":"2024-02-26","arxiv_id":"2402.16664","repositories_listed":1,"syntology":null},{"url":"/paper/bridging-the-gap-between-2d-and-3d-visual","slug":"bridging-the-gap-between-2d-and-3d-visual","title":"Bridging the Gap between 2D and 3D Visual Question Answering: A Fusion Approach for 3D VQA","date":"2024-02-24","arxiv_id":"2402.15933","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":8,"n_instrument":4,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":13,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bridging-the-gap-between-2d-and-3d-visual#ran","syntology_url":"https://syntology.ai/paper/2402.15933","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.15933"}},"official":{"repos":["matthewdm0816/bridgeqa"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/commvqa-situating-visual-question-answering","slug":"commvqa-situating-visual-question-answering","title":"CommVQA: Situating Visual Question Answering in Communicative Contexts","date":"2024-02-22","arxiv_id":"2402.15002","repositories_listed":1,"syntology":null},{"url":"/paper/uncertainty-aware-evaluation-for-vision","slug":"uncertainty-aware-evaluation-for-vision","title":"Uncertainty-Aware Evaluation for Vision-Language Models","date":"2024-02-22","arxiv_id":"2402.14418","repositories_listed":1,"syntology":{"n":17,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":9,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/uncertainty-aware-evaluation-for-vision#ran","syntology_url":"https://syntology.ai/paper/2402.14418","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14418"}},"official":{"repos":["ensec-ai/vlm-uncertainty-bench"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-hallucinations-of-multi-modal-large","slug":"visual-hallucinations-of-multi-modal-large","title":"Visual Hallucinations of Multi-modal Large Language Models","date":"2024-02-22","arxiv_id":"2402.14683","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/visual-hallucinations-of-multi-modal-large#ran","syntology_url":"https://syntology.ai/paper/2402.14683","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14683"}},"official":{"repos":["wenhuang2000/vhtest"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cognitive-visual-language-mapper-advancing","slug":"cognitive-visual-language-mapper-advancing","title":"Cognitive Visual-Language Mapper: Advancing Multimodal Comprehension with Enhanced Visual Knowledge Alignment","date":"2024-02-21","arxiv_id":"2402.13561","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/cognitive-visual-language-mapper-advancing#ran","syntology_url":"https://syntology.ai/paper/2402.13561","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.13561"}},"official":{"repos":["hitsz-tmg/cognitive-visual-language-mapper"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/aligning-modalities-in-vision-large-language","slug":"aligning-modalities-in-vision-large-language","title":"Aligning Modalities in Vision Large Language Models via Preference Fine-tuning","date":"2024-02-18","arxiv_id":"2402.11411","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":4,"n_instrument":4,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":2,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 1 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/aligning-modalities-in-vision-large-language#ran","syntology_url":"https://syntology.ai/paper/2402.11411","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11411"}},"official":{"repos":["yiyangzhou/povid"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/allava-harnessing-gpt4v-synthesized-data-for","slug":"allava-harnessing-gpt4v-synthesized-data-for","title":"ALLaVA: Harnessing GPT4V-Synthesized Data for Lite Vision-Language Models","date":"2024-02-18","arxiv_id":"2402.11684","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/allava-harnessing-gpt4v-synthesized-data-for#ran","syntology_url":"https://syntology.ai/paper/2402.11684","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11684"}},"official":{"repos":["freedomintelligence/allava"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/collavo-crayon-large-language-and-vision","slug":"collavo-crayon-large-language-and-vision","title":"CoLLaVO: Crayon Large Language and Vision mOdel","date":"2024-02-17","arxiv_id":"2402.11248","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/collavo-crayon-large-language-and-vision#ran","syntology_url":"https://syntology.ai/paper/2402.11248","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11248"}},"official":{"repos":["ByungKwanLee/CoLLaVO"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ii-mmr-identifying-and-improving-multi-modal","slug":"ii-mmr-identifying-and-improving-multi-modal","title":"II-MMR: Identifying and Improving Multi-modal Multi-hop Reasoning in Visual Question Answering","date":"2024-02-16","arxiv_id":"2402.11058","repositories_listed":1,"syntology":null},{"url":"/paper/omnimedvqa-a-new-large-scale-comprehensive","slug":"omnimedvqa-a-new-large-scale-comprehensive","title":"OmniMedVQA: A New Large-Scale Comprehensive Evaluation Benchmark for Medical LVLM","date":"2024-02-14","arxiv_id":"2402.09181","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/omnimedvqa-a-new-large-scale-comprehensive#ran","syntology_url":"https://syntology.ai/paper/2402.09181","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.09181"}},"official":{"repos":["opengvlab/multi-modality-arena"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/pretraining-vision-language-model-for","slug":"pretraining-vision-language-model-for","title":"Pretraining Vision-Language Model for Difference Visual Question Answering in Longitudinal Chest X-rays","date":"2024-02-14","arxiv_id":"2402.08966","repositories_listed":1,"syntology":null},{"url":"/paper/preflmr-scaling-up-fine-grained-late","slug":"preflmr-scaling-up-fine-grained-late","title":"PreFLMR: Scaling Up Fine-Grained Late-Interaction Multi-modal Retrievers","date":"2024-02-13","arxiv_id":"2402.08327","repositories_listed":1,"syntology":null},{"url":"/paper/visually-dehallucinative-instruction","slug":"visually-dehallucinative-instruction","title":"Visually Dehallucinative Instruction Generation","date":"2024-02-13","arxiv_id":"2402.08348","repositories_listed":1,"syntology":null},{"url":"/paper/synthesizing-sentiment-controlled-feedback","slug":"synthesizing-sentiment-controlled-feedback","title":"Synthesizing Sentiment-Controlled Feedback For Multimodal Text and Image Data","date":"2024-02-12","arxiv_id":"2402.07640","repositories_listed":1,"syntology":null},{"url":"/paper/a-benchmark-for-multi-modal-foundation-models","slug":"a-benchmark-for-multi-modal-foundation-models","title":"Q-Bench+: A Benchmark for Multi-modal Foundation Models on Low-level Vision from Single Images to Pairs","date":"2024-02-11","arxiv_id":"2402.07116","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-benchmark-for-multi-modal-foundation-models#ran","syntology_url":"https://syntology.ai/paper/2402.07116","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.07116"}},"official":{"repos":["Q-Future/Q-Bench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/open-ended-vqa-benchmarking-of-vision","slug":"open-ended-vqa-benchmarking-of-vision","title":"Open-ended VQA benchmarking of Vision-Language models by exploiting Classification datasets and their semantic hierarchy","date":"2024-02-11","arxiv_id":"2402.07270","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/open-ended-vqa-benchmarking-of-vision#ran","syntology_url":"https://syntology.ai/paper/2402.07270","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.07270"}},"official":{"repos":["lmb-freiburg/ovqa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/gemini-goes-to-med-school-exploring-the","slug":"gemini-goes-to-med-school-exploring-the","title":"Gemini Goes to Med School: Exploring the Capabilities of Multimodal Large Language Models on Medical Challenge Problems & Hallucinations","date":"2024-02-10","arxiv_id":"2402.07023","repositories_listed":1,"syntology":null},{"url":"/paper/examining-gender-and-racial-bias-in-large","slug":"examining-gender-and-racial-bias-in-large","title":"Examining Gender and Racial Bias in Large Vision-Language Models Using a Novel Dataset of Parallel Images","date":"2024-02-08","arxiv_id":"2402.05779","repositories_listed":1,"syntology":null},{"url":"/paper/sphinx-x-scaling-data-and-parameters-for-a","slug":"sphinx-x-scaling-data-and-parameters-for-a","title":"SPHINX-X: Scaling Data and Parameters for a Family of Multi-modal Large Language Models","date":"2024-02-08","arxiv_id":"2402.05935","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sphinx-x-scaling-data-and-parameters-for-a#ran","syntology_url":"https://syntology.ai/paper/2402.05935","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05935"}},"official":{"repos":["alpha-vllm/llama2-accessory"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/convincing-rationales-for-visual-question","slug":"convincing-rationales-for-visual-question","title":"Convincing Rationales for Visual Question Answering Reasoning","date":"2024-02-06","arxiv_id":"2402.03896","repositories_listed":1,"syntology":null},{"url":"/paper/text-guided-image-clustering","slug":"text-guided-image-clustering","title":"Text-Guided Image Clustering","date":"2024-02-05","arxiv_id":"2402.02996","repositories_listed":1,"syntology":null},{"url":"/paper/video-lavit-unified-video-language-pre","slug":"video-lavit-unified-video-language-pre","title":"Video-LaVIT: Unified Video-Language Pre-training with Decoupled Visual-Motional Tokenization","date":"2024-02-05","arxiv_id":"2402.03161","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/video-lavit-unified-video-language-pre#ran","syntology_url":"https://syntology.ai/paper/2402.03161","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.03161"}},"official":null}},{"url":"/paper/gerea-question-aware-prompt-captions-for","slug":"gerea-question-aware-prompt-captions-for","title":"GeReA: Question-Aware Prompt Captions for Knowledge-based Visual Question Answering","date":"2024-02-04","arxiv_id":"2402.02503","repositories_listed":1,"syntology":{"n":18,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":18,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/gerea-question-aware-prompt-captions-for#ran","syntology_url":"https://syntology.ai/paper/2402.02503","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.02503"}},"official":{"repos":["upper9527/gerea"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/knowledge-generation-for-zero-shot-knowledge","slug":"knowledge-generation-for-zero-shot-knowledge","title":"Knowledge Generation for Zero-shot Knowledge-based VQA","date":"2024-02-04","arxiv_id":"2402.02541","repositories_listed":1,"syntology":null},{"url":"/paper/instruction-makes-a-difference","slug":"instruction-makes-a-difference","title":"Instruction Makes a Difference","date":"2024-02-01","arxiv_id":"2402.00453","repositories_listed":1,"syntology":null},{"url":"/paper/mousi-poly-visual-expert-vision-language","slug":"mousi-poly-visual-expert-vision-language","title":"MouSi: Poly-Visual-Expert Vision-Language Models","date":"2024-01-30","arxiv_id":"2401.17221","repositories_listed":1,"syntology":null},{"url":"/paper/internlm-xcomposer2-mastering-free-form-text","slug":"internlm-xcomposer2-mastering-free-form-text","title":"InternLM-XComposer2: Mastering Free-form Text-Image Composition and Comprehension in Vision-Language Large Model","date":"2024-01-29","arxiv_id":"2401.16420","repositories_listed":1,"syntology":null},{"url":"/paper/q-a-prompts-discovering-rich-visual-clues","slug":"q-a-prompts-discovering-rich-visual-clues","title":"Q&A Prompts: Discovering Rich Visual Clues through Mining Question-Answer Prompts for VQA requiring Diverse World Knowledge","date":"2024-01-19","arxiv_id":"2401.10712","repositories_listed":1,"syntology":null},{"url":"/paper/question-answer-cross-language-image-matching","slug":"question-answer-cross-language-image-matching","title":"Question-Answer Cross Language Image Matching for Weakly Supervised Semantic Segmentation","date":"2024-01-18","arxiv_id":"2401.09883","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/question-answer-cross-language-image-matching#ran","syntology_url":"https://syntology.ai/paper/2401.09883","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.09883"}},"official":{"repos":["cvi-szu/qa-clims"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/veagle-advancements-in-multimodal","slug":"veagle-advancements-in-multimodal","title":"Veagle: Advancements in Multimodal Representation Learning","date":"2024-01-18","arxiv_id":"2403.08773","repositories_listed":1,"syntology":null},{"url":"/paper/uncovering-the-full-potential-of-visual","slug":"uncovering-the-full-potential-of-visual","title":"Uncovering the Full Potential of Visual Grounding Methods in VQA","date":"2024-01-15","arxiv_id":"2401.07803","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/uncovering-the-full-potential-of-visual#ran","syntology_url":"https://syntology.ai/paper/2401.07803","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.07803"}},"official":{"repos":["dreichcsl/truevg"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/generalizing-visual-question-answering-from","slug":"generalizing-visual-question-answering-from","title":"Generalizing Visual Question Answering from Synthetic to Human-Written Questions via a Chain of QA with a Large Language Model","date":"2024-01-12","arxiv_id":"2401.06400","repositories_listed":1,"syntology":null},{"url":"/paper/cross-modal-retrieval-for-knowledge-based","slug":"cross-modal-retrieval-for-knowledge-based","title":"Cross-modal Retrieval for Knowledge-based Visual Question Answering","date":"2024-01-11","arxiv_id":"2401.05736","repositories_listed":1,"syntology":{"n":11,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":11,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/cross-modal-retrieval-for-knowledge-based#ran","syntology_url":"https://syntology.ai/paper/2401.05736","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.05736"}},"official":{"repos":["paullerner/viquae"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/hallucination-benchmark-in-medical-visual","slug":"hallucination-benchmark-in-medical-visual","title":"Hallucination Benchmark in Medical Visual Question Answering","date":"2024-01-11","arxiv_id":"2401.05827","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hallucination-benchmark-in-medical-visual#ran","syntology_url":"https://syntology.ai/paper/2401.05827","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.05827"}},"official":{"repos":["knowlab/halt-medvqa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/miss-a-generative-pretraining-and-finetuning","slug":"miss-a-generative-pretraining-and-finetuning","title":"MISS: A Generative Pretraining and Finetuning Approach for Med-VQA","date":"2024-01-10","arxiv_id":"2401.05163","repositories_listed":1,"syntology":null},{"url":"/paper/camml-context-aware-multimodal-learner-for","slug":"camml-context-aware-multimodal-learner-for","title":"CaMML: Context-Aware Multimodal Learner for Large Models","date":"2024-01-06","arxiv_id":"2401.03149","repositories_listed":1,"syntology":null},{"url":"/paper/pefomed-parameter-efficient-fine-tuning-on","slug":"pefomed-parameter-efficient-fine-tuning-on","title":"PeFoMed: Parameter Efficient Fine-tuning of Multimodal Large Language Models for Medical Imaging","date":"2024-01-05","arxiv_id":"2401.02797","repositories_listed":1,"syntology":null},{"url":"/paper/artquest-countering-hidden-language-biases-in","slug":"artquest-countering-hidden-language-biases-in","title":"ArtQuest: Countering Hidden Language Biases in ArtVQA","date":"2024-01-04","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/llava-ph-efficient-multi-modal-assistant-with","slug":"llava-ph-efficient-multi-modal-assistant-with","title":"LLaVA-Phi: Efficient Multi-Modal Assistant with Small Language Model","date":"2024-01-04","arxiv_id":"2401.02330","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":9,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/llava-ph-efficient-multi-modal-assistant-with#ran","syntology_url":"https://syntology.ai/paper/2401.02330","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.02330"}},"official":{"repos":["zhuyiche/llava-phi"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/gpt-4v-ision-is-a-generalist-web-agent-if","slug":"gpt-4v-ision-is-a-generalist-web-agent-if","title":"GPT-4V(ision) is a Generalist Web Agent, if Grounded","date":"2024-01-03","arxiv_id":"2401.01614","repositories_listed":1,"syntology":{"n":14,"n_ran":13,"n_constructed":0,"n_ran_checked":12,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":14,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/gpt-4v-ision-is-a-generalist-web-agent-if#ran","syntology_url":"https://syntology.ai/paper/2401.01614","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.01614"}},"official":{"repos":["osu-nlp-group/seeact"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-a-unified-multimodal-reasoning","slug":"towards-a-unified-multimodal-reasoning","title":"Towards a Unified Multimodal Reasoning Framework","date":"2023-12-22","arxiv_id":"2312.15021","repositories_listed":1,"syntology":null},{"url":"/paper/textit-v-guided-visual-search-as-a-core","slug":"textit-v-guided-visual-search-as-a-core","title":"V*: Guided Visual Search as a Core Mechanism in Multimodal LLMs","date":"2023-12-21","arxiv_id":"2312.14135","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":2,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/textit-v-guided-visual-search-as-a-core#ran","syntology_url":"https://syntology.ai/paper/2312.14135","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.14135"}},"official":{"repos":["penghao-wu/vstar"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vcoder-versatile-vision-encoders-for","slug":"vcoder-versatile-vision-encoders-for","title":"VCoder: Versatile Vision Encoders for Multimodal Large Language Models","date":"2023-12-21","arxiv_id":"2312.14233","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vcoder-versatile-vision-encoders-for#ran","syntology_url":"https://syntology.ai/paper/2312.14233","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.14233"}},"official":{"repos":["shi-labs/vcoder"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/generative-multimodal-models-are-in-context","slug":"generative-multimodal-models-are-in-context","title":"Generative Multimodal Models are In-Context Learners","date":"2023-12-20","arxiv_id":"2312.13286","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/generative-multimodal-models-are-in-context#ran","syntology_url":"https://syntology.ai/paper/2312.13286","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.13286"}},"official":{"repos":["baaivision/emu"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/object-attribute-matters-in-visual-question","slug":"object-attribute-matters-in-visual-question","title":"Object Attribute Matters in Visual Question Answering","date":"2023-12-20","arxiv_id":"2401.09442","repositories_listed":1,"syntology":null},{"url":"/paper/object-aware-adaptive-positivity-learning-for","slug":"object-aware-adaptive-positivity-learning-for","title":"Object-aware Adaptive-Positivity Learning for Audio-Visual Question Answering","date":"2023-12-20","arxiv_id":"2312.12816","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/object-aware-adaptive-positivity-learning-for#ran","syntology_url":"https://syntology.ai/paper/2312.12816","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.12816"}},"official":{"repos":["zhangbin-ai/apl"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/earthvqa-towards-queryable-earth-via","slug":"earthvqa-towards-queryable-earth-via","title":"EarthVQA: Towards Queryable Earth via Relational Reasoning-Based Remote Sensing Visual Question Answering","date":"2023-12-19","arxiv_id":"2312.12222","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/earthvqa-towards-queryable-earth-via#ran","syntology_url":"https://syntology.ai/paper/2312.12222","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.12222"}},"official":{"repos":["Junjue-Wang/EarthVQA"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/gemini-a-family-of-highly-capable-multimodal-1","slug":"gemini-a-family-of-highly-capable-multimodal-1","title":"Gemini: A Family of Highly Capable Multimodal Models","date":"2023-12-19","arxiv_id":"2312.11805","repositories_listed":1,"syntology":null},{"url":"/paper/vqa4cir-boosting-composed-image-retrieval","slug":"vqa4cir-boosting-composed-image-retrieval","title":"VQA4CIR: Boosting Composed Image Retrieval with Visual Question Answering","date":"2023-12-19","arxiv_id":"2312.12273","repositories_listed":1,"syntology":null},{"url":"/paper/haar-text-conditioned-generative-model-of-3d","slug":"haar-text-conditioned-generative-model-of-3d","title":"HAAR: Text-Conditioned Generative Model of 3D Strand-based Human Hairstyles","date":"2023-12-18","arxiv_id":"2312.11666","repositories_listed":1,"syntology":null},{"url":"/paper/osmlocator-locating-overlapping-scatter-marks","slug":"osmlocator-locating-overlapping-scatter-marks","title":"OsmLocator: locating overlapping scatter marks with a non-training generative perspective","date":"2023-12-18","arxiv_id":"2312.11146","repositories_listed":1,"syntology":null},{"url":"/paper/p-laplacian-adaptation-for-generative-pre","slug":"p-laplacian-adaptation-for-generative-pre","title":"p-Laplacian Adaptation for Generative Pre-trained Vision-Language Models","date":"2023-12-17","arxiv_id":"2312.10613","repositories_listed":1,"syntology":null},{"url":"/paper/privacy-aware-document-visual-question","slug":"privacy-aware-document-visual-question","title":"Privacy-Aware Document Visual Question Answering","date":"2023-12-15","arxiv_id":"2312.10108","repositories_listed":1,"syntology":null},{"url":"/paper/skysense-a-multi-modal-remote-sensing","slug":"skysense-a-multi-modal-remote-sensing","title":"SkySense: A Multi-Modal Remote Sensing Foundation Model Towards Universal Interpretation for Earth Observation Imagery","date":"2023-12-15","arxiv_id":"2312.10115","repositories_listed":1,"syntology":null},{"url":"/paper/wordscape-a-pipeline-to-extract-multilingual-1","slug":"wordscape-a-pipeline-to-extract-multilingual-1","title":"WordScape: a Pipeline to extract multilingual, visually rich Documents with Layout Annotations from Web Crawl Data","date":"2023-12-15","arxiv_id":"2312.10188","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/wordscape-a-pipeline-to-extract-multilingual-1#ran","syntology_url":"https://syntology.ai/paper/2312.10188","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.10188"}},"official":{"repos":["DS3Lab/WordScape"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vl-gpt-a-generative-pre-trained-transformer","slug":"vl-gpt-a-generative-pre-trained-transformer","title":"VL-GPT: A Generative Pre-trained Transformer for Vision and Language Understanding and Generation","date":"2023-12-14","arxiv_id":"2312.09251","repositories_listed":1,"syntology":null},{"url":"/paper/vlap-efficient-video-language-alignment-via","slug":"vlap-efficient-video-language-alignment-via","title":"ViLA: Efficient Video-Language Alignment for Video Question Answering","date":"2023-12-13","arxiv_id":"2312.08367","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vlap-efficient-video-language-alignment-via#ran","syntology_url":"https://syntology.ai/paper/2312.08367","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.08367"}},"official":{"repos":["xijun-cs/vila"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hallucination-augmented-contrastive-learning","slug":"hallucination-augmented-contrastive-learning","title":"Hallucination Augmented Contrastive Learning for Multimodal Large Language Model","date":"2023-12-12","arxiv_id":"2312.06968","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/hallucination-augmented-contrastive-learning#ran","syntology_url":"https://syntology.ai/paper/2312.06968","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.06968"}},"official":{"repos":["x-plug/mplug-halowl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/image-content-generation-with-causal","slug":"image-content-generation-with-causal","title":"Image Content Generation with Causal Reasoning","date":"2023-12-12","arxiv_id":"2312.07132","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/image-content-generation-with-causal#ran","syntology_url":"https://syntology.ai/paper/2312.07132","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.07132"}},"official":{"repos":["ieit-agi/mix-shannon"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/genixer-empowering-multimodal-large-language","slug":"genixer-empowering-multimodal-large-language","title":"Genixer: Empowering Multimodal Large Language Models as a Powerful Data Generator","date":"2023-12-11","arxiv_id":"2312.06731","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":4,"n_pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 1 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/genixer-empowering-multimodal-large-language#ran","syntology_url":"https://syntology.ai/paper/2312.06731","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.06731"}},"official":{"repos":["zhaohengyuan1/genixer"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/nuscenes-mqa-integrated-evaluation-of","slug":"nuscenes-mqa-integrated-evaluation-of","title":"NuScenes-MQA: Integrated Evaluation of Captions and QA for Autonomous Driving Datasets using Markup Annotations","date":"2023-12-11","arxiv_id":"2312.06352","repositories_listed":1,"syntology":null},{"url":"/paper/vary-scaling-up-the-vision-vocabulary-for","slug":"vary-scaling-up-the-vision-vocabulary-for","title":"Vary: Scaling up the Vision Vocabulary for Large Vision-Language Models","date":"2023-12-11","arxiv_id":"2312.06109","repositories_listed":1,"syntology":null}],"record_sha256":"36083e466fe1f87cc8f648a99730e2a723f4dcdc6d3ee45989fbe408ed243e95","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}