{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/visual-question-answering/papers/ran/2","list_of":"/task/visual-question-answering","task":"Visual Question Answering (VQA)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":2,"pages_in_order":4,"rows_per_page":100,"rows":[101,200],"of":359,"counts":{"archive_papers_tagged":2167,"with_a_code_link":1039,"where_syntology_ran_a_sample":359,"not_listed_spam_title":0,"listed":2167,"listed_where_code_ran":359,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":287,"every_run_a_failure_of_syntologys_instrument":72,"listed_with_a_run_with_no_instrument_failure":287,"listed_every_run_a_failure_of_syntologys_instrument":72,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/visual-question-answering/papers/ran/1","prev":"/task/visual-question-answering/papers/ran/1","next":"/task/visual-question-answering/papers/ran/3","papers":[{"url":"/paper/vqattack-transferable-adversarial-attacks-on","slug":"vqattack-transferable-adversarial-attacks-on","title":"VQAttack: Transferable Adversarial Attacks on Visual Question Answering via Pre-trained Models","date":"2024-02-16","arxiv_id":"2402.11083","repositories_listed":0,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vqattack-transferable-adversarial-attacks-on#ran","syntology_url":"https://syntology.ai/paper/2402.11083","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11083"}},"official":null}},{"url":"/paper/omnimedvqa-a-new-large-scale-comprehensive","slug":"omnimedvqa-a-new-large-scale-comprehensive","title":"OmniMedVQA: A New Large-Scale Comprehensive Evaluation Benchmark for Medical LVLM","date":"2024-02-14","arxiv_id":"2402.09181","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/omnimedvqa-a-new-large-scale-comprehensive#ran","syntology_url":"https://syntology.ai/paper/2402.09181","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.09181"}},"official":{"repos":["opengvlab/multi-modality-arena"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/kvq-kaleidoscope-video-quality-assessment-for","slug":"kvq-kaleidoscope-video-quality-assessment-for","title":"KVQ: Kwai Video Quality Assessment for Short-form Videos","date":"2024-02-11","arxiv_id":"2402.07220","repositories_listed":1,"syntology":{"n":28,"n_ran":22,"n_constructed":10,"n_ran_checked":15,"n_instrument":7,"n_unverified":6,"n_honours":1,"n_violates":0,"n_no_contract":14,"n_pointer_only":28,"phrase":"22 ran (of which 10 constructed an object rather than computing a result; 15 with no instrument failure: 1 honoured, 0 violated, 14 with no contract checked; 7 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/kvq-kaleidoscope-video-quality-assessment-for#ran","syntology_url":"https://syntology.ai/paper/2402.07220","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.07220"}},"official":null}},{"url":"/paper/open-ended-vqa-benchmarking-of-vision","slug":"open-ended-vqa-benchmarking-of-vision","title":"Open-ended VQA benchmarking of Vision-Language models by exploiting Classification datasets and their semantic hierarchy","date":"2024-02-11","arxiv_id":"2402.07270","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/open-ended-vqa-benchmarking-of-vision#ran","syntology_url":"https://syntology.ai/paper/2402.07270","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.07270"}},"official":{"repos":["lmb-freiburg/ovqa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/video-lavit-unified-video-language-pre","slug":"video-lavit-unified-video-language-pre","title":"Video-LaVIT: Unified Video-Language Pre-training with Decoupled Visual-Motional Tokenization","date":"2024-02-05","arxiv_id":"2402.03161","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/video-lavit-unified-video-language-pre#ran","syntology_url":"https://syntology.ai/paper/2402.03161","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.03161"}},"official":null}},{"url":"/paper/gerea-question-aware-prompt-captions-for","slug":"gerea-question-aware-prompt-captions-for","title":"GeReA: Question-Aware Prompt Captions for Knowledge-based Visual Question Answering","date":"2024-02-04","arxiv_id":"2402.02503","repositories_listed":1,"syntology":{"n":18,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":18,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/gerea-question-aware-prompt-captions-for#ran","syntology_url":"https://syntology.ai/paper/2402.02503","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.02503"}},"official":{"repos":["upper9527/gerea"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/question-answer-cross-language-image-matching","slug":"question-answer-cross-language-image-matching","title":"Question-Answer Cross Language Image Matching for Weakly Supervised Semantic Segmentation","date":"2024-01-18","arxiv_id":"2401.09883","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/question-answer-cross-language-image-matching#ran","syntology_url":"https://syntology.ai/paper/2401.09883","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.09883"}},"official":{"repos":["cvi-szu/qa-clims"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/uncovering-the-full-potential-of-visual","slug":"uncovering-the-full-potential-of-visual","title":"Uncovering the Full Potential of Visual Grounding Methods in VQA","date":"2024-01-15","arxiv_id":"2401.07803","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/uncovering-the-full-potential-of-visual#ran","syntology_url":"https://syntology.ai/paper/2401.07803","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.07803"}},"official":{"repos":["dreichcsl/truevg"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cross-modal-retrieval-for-knowledge-based","slug":"cross-modal-retrieval-for-knowledge-based","title":"Cross-modal Retrieval for Knowledge-based Visual Question Answering","date":"2024-01-11","arxiv_id":"2401.05736","repositories_listed":1,"syntology":{"n":11,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":11,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/cross-modal-retrieval-for-knowledge-based#ran","syntology_url":"https://syntology.ai/paper/2401.05736","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.05736"}},"official":{"repos":["paullerner/viquae"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/hallucination-benchmark-in-medical-visual","slug":"hallucination-benchmark-in-medical-visual","title":"Hallucination Benchmark in Medical Visual Question Answering","date":"2024-01-11","arxiv_id":"2401.05827","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hallucination-benchmark-in-medical-visual#ran","syntology_url":"https://syntology.ai/paper/2401.05827","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.05827"}},"official":{"repos":["knowlab/halt-medvqa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/3dmit-3d-multi-modal-instruction-tuning-for","slug":"3dmit-3d-multi-modal-instruction-tuning-for","title":"3DMIT: 3D Multi-modal Instruction Tuning for Scene Understanding","date":"2024-01-06","arxiv_id":"2401.03201","repositories_listed":1,"syntology":{"n":10,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":10,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/3dmit-3d-multi-modal-instruction-tuning-for#ran","syntology_url":"https://syntology.ai/paper/2401.03201","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.03201"}},"official":{"repos":["staymylove/3DMIT"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/mining-fine-grained-image-text-alignment-for","slug":"mining-fine-grained-image-text-alignment-for","title":"Mining Fine-Grained Image-Text Alignment for Zero-Shot Captioning via Text-Only Training","date":"2024-01-04","arxiv_id":"2401.02347","repositories_listed":1,"syntology":{"n":8,"n_ran":3,"n_constructed":1,"n_ran_checked":1,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":8,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/mining-fine-grained-image-text-alignment-for#ran","syntology_url":"https://syntology.ai/paper/2401.02347","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.02347"}},"official":{"repos":["artanic30/maccap"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":5,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/tinygpt-v-efficient-multimodal-large-language","slug":"tinygpt-v-efficient-multimodal-large-language","title":"TinyGPT-V: Efficient Multimodal Large Language Model via Small Backbones","date":"2023-12-28","arxiv_id":"2312.16862","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":1,"n_instrument":6,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tinygpt-v-efficient-multimodal-large-language#ran","syntology_url":"https://syntology.ai/paper/2312.16862","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.16862"}},"official":{"repos":["dlyuangod/tinygpt-v"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/q-align-teaching-lmms-for-visual-scoring-via","slug":"q-align-teaching-lmms-for-visual-scoring-via","title":"Q-Align: Teaching LMMs for Visual Scoring via Discrete Text-Defined Levels","date":"2023-12-28","arxiv_id":"2312.17090","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/q-align-teaching-lmms-for-visual-scoring-via#ran","syntology_url":"https://syntology.ai/paper/2312.17090","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.17090"}},"official":{"repos":["q-future/q-align"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/internvl-scaling-up-vision-foundation-models","slug":"internvl-scaling-up-vision-foundation-models","title":"InternVL: Scaling up Vision Foundation Models and Aligning for Generic Visual-Linguistic Tasks","date":"2023-12-21","arxiv_id":"2312.14238","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/internvl-scaling-up-vision-foundation-models#ran","syntology_url":"https://syntology.ai/paper/2312.14238","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.14238"}},"official":{"repos":["opengvlab/internvl"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/earthvqa-towards-queryable-earth-via","slug":"earthvqa-towards-queryable-earth-via","title":"EarthVQA: Towards Queryable Earth via Relational Reasoning-Based Remote Sensing Visual Question Answering","date":"2023-12-19","arxiv_id":"2312.12222","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/earthvqa-towards-queryable-earth-via#ran","syntology_url":"https://syntology.ai/paper/2312.12222","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.12222"}},"official":{"repos":["Junjue-Wang/EarthVQA"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/advancing-surgical-vqa-with-scene-graph","slug":"advancing-surgical-vqa-with-scene-graph","title":"Advancing Surgical VQA with Scene Graph Knowledge","date":"2023-12-15","arxiv_id":"2312.10251","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/advancing-surgical-vqa-with-scene-graph#ran","syntology_url":"https://syntology.ai/paper/2312.10251","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.10251"}},"official":{"repos":["camma-public/ssg-qa","camma-public/ssg-vqa"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cogagent-a-visual-language-model-for-gui","slug":"cogagent-a-visual-language-model-for-gui","title":"CogAgent: A Visual Language Model for GUI Agents","date":"2023-12-14","arxiv_id":"2312.08914","repositories_listed":3,"syntology":{"n":18,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":1,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/cogagent-a-visual-language-model-for-gui#ran","syntology_url":"https://syntology.ai/paper/2312.08914","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.08914"}},"official":{"repos":["THUDM/CogAgent","thudm/cogvlm"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/vlap-efficient-video-language-alignment-via","slug":"vlap-efficient-video-language-alignment-via","title":"ViLA: Efficient Video-Language Alignment for Video Question Answering","date":"2023-12-13","arxiv_id":"2312.08367","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vlap-efficient-video-language-alignment-via#ran","syntology_url":"https://syntology.ai/paper/2312.08367","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.08367"}},"official":{"repos":["xijun-cs/vila"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/genixer-empowering-multimodal-large-language","slug":"genixer-empowering-multimodal-large-language","title":"Genixer: Empowering Multimodal Large Language Models as a Powerful Data Generator","date":"2023-12-11","arxiv_id":"2312.06731","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":4,"n_pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 1 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/genixer-empowering-multimodal-large-language#ran","syntology_url":"https://syntology.ai/paper/2312.06731","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.06731"}},"official":{"repos":["zhaohengyuan1/genixer"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/quilt-llava-visual-instruction-tuning-by","slug":"quilt-llava-visual-instruction-tuning-by","title":"Quilt-LLaVA: Visual Instruction Tuning by Extracting Localized Narratives from Open-Source Histopathology Videos","date":"2023-12-07","arxiv_id":"2312.04746","repositories_listed":2,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/quilt-llava-visual-instruction-tuning-by#ran","syntology_url":"https://syntology.ai/paper/2312.04746","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.04746"}},"official":null}},{"url":"/paper/language-informed-visual-concept-learning","slug":"language-informed-visual-concept-learning","title":"Language-Informed Visual Concept Learning","date":"2023-12-06","arxiv_id":"2312.03587","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":10,"n_pointer_only":13,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/language-informed-visual-concept-learning#ran","syntology_url":"https://syntology.ai/paper/2312.03587","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.03587"}},"official":{"repos":["sharonal10/langint"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/how-to-configure-good-in-context-sequence-for","slug":"how-to-configure-good-in-context-sequence-for","title":"How to Configure Good In-Context Sequence for Visual Question Answering","date":"2023-12-04","arxiv_id":"2312.01571","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/how-to-configure-good-in-context-sequence-for#ran","syntology_url":"https://syntology.ai/paper/2312.01571","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.01571"}},"official":{"repos":["garyjiajia/ofv2_icl_vqa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/recursive-visual-programming","slug":"recursive-visual-programming","title":"Recursive Visual Programming","date":"2023-12-04","arxiv_id":"2312.02249","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":13,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/recursive-visual-programming#ran","syntology_url":"https://syntology.ai/paper/2312.02249","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.02249"}},"official":{"repos":["para-lost/rvp"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/how-many-unicorns-are-in-this-image-a-safety","slug":"how-many-unicorns-are-in-this-image-a-safety","title":"How Many Unicorns Are in This Image? A Safety Evaluation Benchmark for Vision LLMs","date":"2023-11-27","arxiv_id":"2311.16101","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/how-many-unicorns-are-in-this-image-a-safety#ran","syntology_url":"https://syntology.ai/paper/2311.16101","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.16101"}},"official":{"repos":["ucsc-vlaa/vllm-safety-benchmark"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/boosting-the-power-of-small-multimodal","slug":"boosting-the-power-of-small-multimodal","title":"Boosting the Power of Small Multimodal Reasoning Models to Match Larger Models with Self-Consistency Training","date":"2023-11-23","arxiv_id":"2311.14109","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/boosting-the-power-of-small-multimodal#ran","syntology_url":"https://syntology.ai/paper/2311.14109","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.14109"}},"official":{"repos":["chengtan9907/mc-cot"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/video-llava-learning-united-visual-1","slug":"video-llava-learning-united-visual-1","title":"Video-LLaVA: Learning United Visual Representation by Alignment Before Projection","date":"2023-11-16","arxiv_id":"2311.10122","repositories_listed":6,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/video-llava-learning-united-visual-1#ran","syntology_url":"https://syntology.ai/paper/2311.10122","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.10122"}},"official":{"repos":["PKU-YuanGroup/Video-LLaVA"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/attribute-diversity-determines-the","slug":"attribute-diversity-determines-the","title":"Attribute Diversity Determines the Systematicity Gap in VQA","date":"2023-11-15","arxiv_id":"2311.08695","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/attribute-diversity-determines-the#ran","syntology_url":"https://syntology.ai/paper/2311.08695","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.08695"}},"official":{"repos":["ikb-a/systematicity-gap-in-vqa"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sphinx-the-joint-mixing-of-weights-tasks-and","slug":"sphinx-the-joint-mixing-of-weights-tasks-and","title":"SPHINX: The Joint Mixing of Weights, Tasks, and Visual Embeddings for Multi-modal Large Language Models","date":"2023-11-13","arxiv_id":"2311.07575","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sphinx-the-joint-mixing-of-weights-tasks-and#ran","syntology_url":"https://syntology.ai/paper/2311.07575","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.07575"}},"official":{"repos":["alpha-vllm/llama2-accessory"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/monkey-image-resolution-and-text-label-are","slug":"monkey-image-resolution-and-text-label-are","title":"Monkey: Image Resolution and Text Label Are Important Things for Large Multi-modal Models","date":"2023-11-11","arxiv_id":"2311.06607","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/monkey-image-resolution-and-text-label-are#ran","syntology_url":"https://syntology.ai/paper/2311.06607","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.06607"}},"official":{"repos":["yuliang-liu/monkey"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/analyzing-modular-approaches-for-visual","slug":"analyzing-modular-approaches-for-visual","title":"Analyzing Modular Approaches for Visual Question Decomposition","date":"2023-11-10","arxiv_id":"2311.06411","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/analyzing-modular-approaches-for-visual#ran","syntology_url":"https://syntology.ai/paper/2311.06411","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.06411"}},"official":{"repos":["brown-palm/visual-question-decomposition"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/mplug-owl2-revolutionizing-multi-modal-large","slug":"mplug-owl2-revolutionizing-multi-modal-large","title":"mPLUG-Owl2: Revolutionizing Multi-modal Large Language Model with Modality Collaboration","date":"2023-11-07","arxiv_id":"2311.04257","repositories_listed":2,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mplug-owl2-revolutionizing-multi-modal-large#ran","syntology_url":"https://syntology.ai/paper/2311.04257","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.04257"}},"official":{"repos":["x-plug/mplug-owl"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/language-guided-visual-question-answering","slug":"language-guided-visual-question-answering","title":"Language Guided Visual Question Answering: Elevate Your Multimodal Language Model Using Knowledge-Enriched Prompts","date":"2023-10-31","arxiv_id":"2310.20159","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/language-guided-visual-question-answering#ran","syntology_url":"https://syntology.ai/paper/2310.20159","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.20159"}},"official":{"repos":["declare-lab/lg-vqa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ehrxqa-a-multi-modal-question-answering-1","slug":"ehrxqa-a-multi-modal-question-answering-1","title":"EHRXQA: A Multi-Modal Question Answering Dataset for Electronic Health Records with Chest X-ray Images","date":"2023-10-28","arxiv_id":"2310.18652","repositories_listed":3,"syntology":{"n":20,"n_ran":20,"n_constructed":0,"n_ran_checked":17,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":17,"n_pointer_only":0,"phrase":"20 ran (of which 0 constructed an object rather than computing a result; 17 with no instrument failure: 0 honoured, 0 violated, 17 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ehrxqa-a-multi-modal-question-answering-1#ran","syntology_url":"https://syntology.ai/paper/2310.18652","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.18652"}},"official":{"repos":["baeseongsu/ehrxqa","baeseongsu/mimic-cxr-vqa"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/3d-aware-visual-question-answering-about-1","slug":"3d-aware-visual-question-answering-about-1","title":"3D-Aware Visual Question Answering about Parts, Poses and Occlusions","date":"2023-10-27","arxiv_id":"2310.17914","repositories_listed":2,"syntology":{"n":28,"n_ran":22,"n_constructed":0,"n_ran_checked":21,"n_instrument":1,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":21,"n_pointer_only":14,"phrase":"22 ran (of which 0 constructed an object rather than computing a result; 21 with no instrument failure: 0 honoured, 0 violated, 21 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/3d-aware-visual-question-answering-about-1#ran","syntology_url":"https://syntology.ai/paper/2310.17914","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.17914"}},"official":{"repos":["xingruiwang/3d-aware-vqa"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/hallusionbench-you-see-what-you-think-or-you","slug":"hallusionbench-you-see-what-you-think-or-you","title":"HallusionBench: An Advanced Diagnostic Suite for Entangled Language Hallucination and Visual Illusion in Large Vision-Language Models","date":"2023-10-23","arxiv_id":"2310.14566","repositories_listed":9,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/hallusionbench-you-see-what-you-think-or-you#ran","syntology_url":"https://syntology.ai/paper/2310.14566","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.14566"}},"official":{"repos":["tianyi-lab/hallusionbench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/a-simple-baseline-for-knowledge-based-visual","slug":"a-simple-baseline-for-knowledge-based-visual","title":"A Simple Baseline for Knowledge-Based Visual Question Answering","date":"2023-10-20","arxiv_id":"2310.13570","repositories_listed":0,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-simple-baseline-for-knowledge-based-visual#ran","syntology_url":"https://syntology.ai/paper/2310.13570","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.13570"}},"official":null}},{"url":"/paper/pali-3-vision-language-models-smaller-faster","slug":"pali-3-vision-language-models-smaller-faster","title":"PaLI-3 Vision Language Models: Smaller, Faster, Stronger","date":"2023-10-13","arxiv_id":"2310.09199","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":3,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 3 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pali-3-vision-language-models-smaller-faster#ran","syntology_url":"https://syntology.ai/paper/2310.09199","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.09199"}},"official":null}},{"url":"/paper/rephrase-augment-reason-visual-grounding-of","slug":"rephrase-augment-reason-visual-grounding-of","title":"Rephrase, Augment, Reason: Visual Grounding of Questions for Vision-Language Models","date":"2023-10-09","arxiv_id":"2310.05861","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rephrase-augment-reason-visual-grounding-of#ran","syntology_url":"https://syntology.ai/paper/2310.05861","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.05861"}},"official":{"repos":["archiki/repare"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/improved-baselines-with-visual-instruction","slug":"improved-baselines-with-visual-instruction","title":"Improved Baselines with Visual Instruction Tuning","date":"2023-10-05","arxiv_id":"2310.03744","repositories_listed":9,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":3,"n_no_contract":0,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 3 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/improved-baselines-with-visual-instruction#ran","syntology_url":"https://syntology.ai/paper/2310.03744","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.03744"}},"official":null}},{"url":"/paper/halle-switch-rethinking-and-controlling","slug":"halle-switch-rethinking-and-controlling","title":"HallE-Control: Controlling Object Hallucination in Large Multimodal Models","date":"2023-10-03","arxiv_id":"2310.01779","repositories_listed":2,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":9,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/halle-switch-rethinking-and-controlling#ran","syntology_url":"https://syntology.ai/paper/2310.01779","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.01779"}},"official":{"repos":["bronyayang/HallE_Switch","bronyayang/halle_control"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-task-performance-evaluating-and","slug":"beyond-task-performance-evaluating-and","title":"Beyond Task Performance: Evaluating and Reducing the Flaws of Large Multimodal Models with In-Context Learning","date":"2023-10-01","arxiv_id":"2310.00647","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":8,"n_pointer_only":11,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 1 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/beyond-task-performance-evaluating-and#ran","syntology_url":"https://syntology.ai/paper/2310.00647","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.00647"}},"official":{"repos":["mshukor/EvALign-ICL"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fine-grained-late-interaction-multi-modal-1","slug":"fine-grained-late-interaction-multi-modal-1","title":"Fine-grained Late-interaction Multi-modal Retrieval for Retrieval Augmented Visual Question Answering","date":"2023-09-29","arxiv_id":"2309.17133","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":3,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fine-grained-late-interaction-multi-modal-1#ran","syntology_url":"https://syntology.ai/paper/2309.17133","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.17133"}},"official":{"repos":["linweizhedragon/retrieval-augmented-visual-question-answering"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vulnerabilities-in-video-quality-assessment-1","slug":"vulnerabilities-in-video-quality-assessment-1","title":"Vulnerabilities in Video Quality Assessment Models: The Challenge of Adversarial Attacks","date":"2023-09-24","arxiv_id":"2309.13609","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vulnerabilities-in-video-quality-assessment-1#ran","syntology_url":"https://syntology.ai/paper/2309.13609","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.13609"}},"official":{"repos":["gzhu-dvl/attackvqa"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/can-i-trust-your-answer-visually-grounded","slug":"can-i-trust-your-answer-visually-grounded","title":"Can I Trust Your Answer? Visually Grounded Video Question Answering","date":"2023-09-04","arxiv_id":"2309.01327","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":1,"n_no_contract":2,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 1 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/can-i-trust-your-answer-visually-grounded#ran","syntology_url":"https://syntology.ai/paper/2309.01327","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.01327"}},"official":{"repos":["doc-doc/next-gqa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/bliva-a-simple-multimodal-llm-for-better","slug":"bliva-a-simple-multimodal-llm-for-better","title":"BLIVA: A Simple Multimodal LLM for Better Handling of Text-Rich Visual Questions","date":"2023-08-19","arxiv_id":"2308.09936","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/bliva-a-simple-multimodal-llm-for-better#ran","syntology_url":"https://syntology.ai/paper/2308.09936","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09936"}},"official":{"repos":["mlpc-ucsd/bliva"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/open-vocabulary-video-question-answering-a","slug":"open-vocabulary-video-question-answering-a","title":"Open-vocabulary Video Question Answering: A New Benchmark for Evaluating the Generalizability of Video Question Answering Models","date":"2023-08-18","arxiv_id":"2308.09363","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":8,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/open-vocabulary-video-question-answering-a#ran","syntology_url":"https://syntology.ai/paper/2308.09363","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09363"}},"official":{"repos":["mlvlab/ovqa"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/vl-pet-vision-and-language-parameter","slug":"vl-pet-vision-and-language-parameter","title":"VL-PET: Vision-and-Language Parameter-Efficient Tuning via Granularity Control","date":"2023-08-18","arxiv_id":"2308.09804","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vl-pet-vision-and-language-parameter#ran","syntology_url":"https://syntology.ai/paper/2308.09804","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09804"}},"official":{"repos":["henryhzy/vl-pet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/uni-nlx-unifying-textual-explanations-for","slug":"uni-nlx-unifying-textual-explanations-for","title":"Uni-NLX: Unifying Textual Explanations for Vision and Vision-Language Tasks","date":"2023-08-17","arxiv_id":"2308.09033","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/uni-nlx-unifying-textual-explanations-for#ran","syntology_url":"https://syntology.ai/paper/2308.09033","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09033"}},"official":{"repos":["fawazsammani/uni-nlx","fawazsammani/nlxgpt"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pro-cap-leveraging-a-frozen-vision-language","slug":"pro-cap-leveraging-a-frozen-vision-language","title":"Pro-Cap: Leveraging a Frozen Vision-Language Model for Hateful Meme Detection","date":"2023-08-16","arxiv_id":"2308.08088","repositories_listed":2,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/pro-cap-leveraging-a-frozen-vision-language#ran","syntology_url":"https://syntology.ai/paper/2308.08088","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.08088"}},"official":{"repos":["social-ai-studio/pro-cap"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/tech-text-guided-reconstruction-of-lifelike","slug":"tech-text-guided-reconstruction-of-lifelike","title":"TeCH: Text-guided Reconstruction of Lifelike Clothed Humans","date":"2023-08-16","arxiv_id":"2308.08545","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":10,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/tech-text-guided-reconstruction-of-lifelike#ran","syntology_url":"https://syntology.ai/paper/2308.08545","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.08545"}},"official":{"repos":["huangyangyi/tech"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/scigraphqa-a-large-scale-synthetic-multi-turn","slug":"scigraphqa-a-large-scale-synthetic-multi-turn","title":"SciGraphQA: A Large-Scale Synthetic Multi-Turn Question-Answering Dataset for Scientific Graphs","date":"2023-08-07","arxiv_id":"2308.03349","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/scigraphqa-a-large-scale-synthetic-multi-turn#ran","syntology_url":"https://syntology.ai/paper/2308.03349","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.03349"}},"official":{"repos":["findalexli/SciGraphQA"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/expert-knowledge-aware-image-difference-graph","slug":"expert-knowledge-aware-image-difference-graph","title":"Expert Knowledge-Aware Image Difference Graph Representation Learning for Difference-Aware Medical Visual Question Answering","date":"2023-07-22","arxiv_id":"2307.11986","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/expert-knowledge-aware-image-difference-graph#ran","syntology_url":"https://syntology.ai/paper/2307.11986","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.11986"}},"official":{"repos":["holipori/mimic-diff-vqa"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/co-attention-gated-vision-language-embedding","slug":"co-attention-gated-vision-language-embedding","title":"CAT-ViL: Co-Attention Gated Vision-Language Embedding for Visual Question Localized-Answering in Robotic Surgery","date":"2023-07-11","arxiv_id":"2307.05182","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":5,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/co-attention-gated-vision-language-embedding#ran","syntology_url":"https://syntology.ai/paper/2307.05182","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.05182"}},"official":{"repos":["longbai1006/cat-vil"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/gpt4roi-instruction-tuning-large-language","slug":"gpt4roi-instruction-tuning-large-language","title":"GPT4RoI: Instruction Tuning Large Language Model on Region-of-Interest","date":"2023-07-07","arxiv_id":"2307.03601","repositories_listed":3,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/gpt4roi-instruction-tuning-large-language#ran","syntology_url":"https://syntology.ai/paper/2307.03601","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.03601"}},"official":{"repos":["jshilong/gpt4roi"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/answer-mining-from-a-pool-of-images-towards","slug":"answer-mining-from-a-pool-of-images-towards","title":"Answer Mining from a Pool of Images: Towards Retrieval-Based Visual Question Answering","date":"2023-06-29","arxiv_id":"2306.16713","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":6,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/answer-mining-from-a-pool-of-images-towards#ran","syntology_url":"https://syntology.ai/paper/2306.16713","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.16713"}},"official":{"repos":["Abhiram4572/mi_bart"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/llavar-enhanced-visual-instruction-tuning-for","slug":"llavar-enhanced-visual-instruction-tuning-for","title":"LLaVAR: Enhanced Visual Instruction Tuning for Text-Rich Image Understanding","date":"2023-06-29","arxiv_id":"2306.17107","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":0,"n_instrument":6,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/llavar-enhanced-visual-instruction-tuning-for#ran","syntology_url":"https://syntology.ai/paper/2306.17107","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.17107"}},"official":{"repos":["SALT-NLP/LLaVAR"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official","unlocated"]}}},{"url":"/paper/shikra-unleashing-multimodal-llm-s","slug":"shikra-unleashing-multimodal-llm-s","title":"Shikra: Unleashing Multimodal LLM's Referential Dialogue Magic","date":"2023-06-27","arxiv_id":"2306.15195","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/shikra-unleashing-multimodal-llm-s#ran","syntology_url":"https://syntology.ai/paper/2306.15195","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.15195"}},"official":{"repos":["shikras/shikra"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/funqa-towards-surprising-video-comprehension","slug":"funqa-towards-surprising-video-comprehension","title":"FunQA: Towards Surprising Video Comprehension","date":"2023-06-26","arxiv_id":"2306.14899","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/funqa-towards-surprising-video-comprehension#ran","syntology_url":"https://syntology.ai/paper/2306.14899","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.14899"}},"official":{"repos":["jingkang50/funqa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/investigating-prompting-techniques-for-zero","slug":"investigating-prompting-techniques-for-zero","title":"Investigating Prompting Techniques for Zero- and Few-Shot Visual Question Answering","date":"2023-06-16","arxiv_id":"2306.09996","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/investigating-prompting-techniques-for-zero#ran","syntology_url":"https://syntology.ai/paper/2306.09996","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.09996"}},"official":{"repos":["rabiulcste/vqazero"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/scalable-neural-probabilistic-answer-set","slug":"scalable-neural-probabilistic-answer-set","title":"Scalable Neural-Probabilistic Answer Set Programming","date":"2023-06-14","arxiv_id":"2306.08397","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scalable-neural-probabilistic-answer-set#ran","syntology_url":"https://syntology.ai/paper/2306.08397","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.08397"}},"official":{"repos":["ml-research/slash"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/layout-and-task-aware-instruction-prompt-for","slug":"layout-and-task-aware-instruction-prompt-for","title":"Layout and Task Aware Instruction Prompt for Zero-shot Document Image Question Answering","date":"2023-06-01","arxiv_id":"2306.00526","repositories_listed":3,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/layout-and-task-aware-instruction-prompt-for#ran","syntology_url":"https://syntology.ai/paper/2306.00526","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.00526"}},"official":{"repos":["wenjinw/latin-prompt"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/vast-a-vision-audio-subtitle-text-omni-1","slug":"vast-a-vision-audio-subtitle-text-omni-1","title":"VAST: A Vision-Audio-Subtitle-Text Omni-Modality Foundation Model and Dataset","date":"2023-05-29","arxiv_id":"2305.18500","repositories_listed":2,"syntology":{"n":42,"n_ran":35,"n_constructed":4,"n_ran_checked":29,"n_instrument":6,"n_unverified":7,"n_honours":2,"n_violates":1,"n_no_contract":26,"n_pointer_only":8,"phrase":"35 ran (of which 4 constructed an object rather than computing a result; 29 with no instrument failure: 2 honoured, 1 violated, 26 with no contract checked; 6 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/vast-a-vision-audio-subtitle-text-omni-1#ran","syntology_url":"https://syntology.ai/paper/2305.18500","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.18500"}},"official":{"repos":["txh-mercury/vast"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":4,"n_ran_no_instrument_failure":12,"n_unverified":7,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/pali-x-on-scaling-up-a-multilingual-vision","slug":"pali-x-on-scaling-up-a-multilingual-vision","title":"PaLI-X: On Scaling up a Multilingual Vision and Language Model","date":"2023-05-29","arxiv_id":"2305.18565","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":4,"n_no_contract":1,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 4 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pali-x-on-scaling-up-a-multilingual-vision#ran","syntology_url":"https://syntology.ai/paper/2305.18565","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.18565"}},"official":null}},{"url":"/paper/crossget-cross-guided-ensemble-of-tokens-for","slug":"crossget-cross-guided-ensemble-of-tokens-for","title":"CrossGET: Cross-Guided Ensemble of Tokens for Accelerating Vision-Language Transformers","date":"2023-05-27","arxiv_id":"2305.17455","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/crossget-cross-guided-ensemble-of-tokens-for#ran","syntology_url":"https://syntology.ai/paper/2305.17455","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.17455"}},"official":{"repos":["sdc17/crossget"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/nuscenes-qa-a-multi-modal-visual-question","slug":"nuscenes-qa-a-multi-modal-visual-question","title":"NuScenes-QA: A Multi-modal Visual Question Answering Benchmark for Autonomous Driving Scenario","date":"2023-05-24","arxiv_id":"2305.14836","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/nuscenes-qa-a-multi-modal-visual-question#ran","syntology_url":"https://syntology.ai/paper/2305.14836","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14836"}},"official":{"repos":["qiantianwen/nuscenes-qa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/surgical-vqla-transformer-with-gated-vision","slug":"surgical-vqla-transformer-with-gated-vision","title":"Surgical-VQLA: Transformer with Gated Vision-Language Embedding for Visual Question Localized-Answering in Robotic Surgery","date":"2023-05-19","arxiv_id":"2305.11692","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":4,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/surgical-vqla-transformer-with-gated-vision#ran","syntology_url":"https://syntology.ai/paper/2305.11692","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.11692"}},"official":{"repos":["longbai1006/surgical-vqla"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/one-peace-exploring-one-general","slug":"one-peace-exploring-one-general","title":"ONE-PEACE: Exploring One General Representation Model Toward Unlimited Modalities","date":"2023-05-18","arxiv_id":"2305.11172","repositories_listed":2,"syntology":{"n":7,"n_ran":5,"n_constructed":2,"n_ran_checked":4,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"5 ran (of which 2 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/one-peace-exploring-one-general#ran","syntology_url":"https://syntology.ai/paper/2305.11172","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.11172"}},"official":{"repos":["OFA-Sys/ONE-PEACE"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":2,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/pmc-vqa-visual-instruction-tuning-for-medical","slug":"pmc-vqa-visual-instruction-tuning-for-medical","title":"PMC-VQA: Visual Instruction Tuning for Medical Visual Question Answering","date":"2023-05-17","arxiv_id":"2305.10415","repositories_listed":2,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pmc-vqa-visual-instruction-tuning-for-medical#ran","syntology_url":"https://syntology.ai/paper/2305.10415","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.10415"}},"official":{"repos":["xiaoman-zhang/PMC-VQA"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/an-empirical-study-of-multimodal-model","slug":"an-empirical-study-of-multimodal-model","title":"An Empirical Study of Multimodal Model Merging","date":"2023-04-28","arxiv_id":"2304.14933","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-empirical-study-of-multimodal-model#ran","syntology_url":"https://syntology.ai/paper/2304.14933","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.14933"}},"official":{"repos":["ylsung/vl-merging"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/surgicalgpt-end-to-end-language-vision-gpt","slug":"surgicalgpt-end-to-end-language-vision-gpt","title":"SurgicalGPT: End-to-End Language-Vision GPT for Visual Question Answering in Surgery","date":"2023-04-19","arxiv_id":"2304.09974","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/surgicalgpt-end-to-end-language-vision-gpt#ran","syntology_url":"https://syntology.ai/paper/2304.09974","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.09974"}},"official":{"repos":["lalithjets/surgicalgpt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-situation-hyper-graphs-for-video","slug":"learning-situation-hyper-graphs-for-video","title":"Learning Situation Hyper-Graphs for Video Question Answering","date":"2023-04-18","arxiv_id":"2304.08682","repositories_listed":1,"syntology":{"n":15,"n_ran":8,"n_constructed":5,"n_ran_checked":8,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":15,"phrase":"8 ran (of which 5 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/learning-situation-hyper-graphs-for-video#ran","syntology_url":"https://syntology.ai/paper/2304.08682","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.08682"}},"official":{"repos":["aurooj/shg-vqa"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":5,"n_ran_no_instrument_failure":8,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/mammut-a-simple-architecture-for-joint","slug":"mammut-a-simple-architecture-for-joint","title":"MaMMUT: A Simple Architecture for Joint Learning for MultiModal Tasks","date":"2023-03-29","arxiv_id":"2303.16839","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":3,"n_no_contract":0,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 3 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mammut-a-simple-architecture-for-joint#ran","syntology_url":"https://syntology.ai/paper/2303.16839","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.16839"}},"official":null}},{"url":"/paper/unmasked-teacher-towards-training-efficient","slug":"unmasked-teacher-towards-training-efficient","title":"Unmasked Teacher: Towards Training-Efficient Video Foundation Models","date":"2023-03-28","arxiv_id":"2303.16058","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":3,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unmasked-teacher-towards-training-efficient#ran","syntology_url":"https://syntology.ai/paper/2303.16058","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.16058"}},"official":{"repos":["opengvlab/unmasked_teacher"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/video-text-as-game-players-hierarchical","slug":"video-text-as-game-players-hierarchical","title":"Video-Text as Game Players: Hierarchical Banzhaf Interaction for Cross-Modal Representation Learning","date":"2023-03-25","arxiv_id":"2303.14369","repositories_listed":4,"syntology":{"n":16,"n_ran":12,"n_constructed":7,"n_ran_checked":11,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"12 ran (of which 7 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/video-text-as-game-players-hierarchical#ran","syntology_url":"https://syntology.ai/paper/2303.14369","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.14369"}},"official":{"repos":["jpthu17/HBI"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/meltr-meta-loss-transformer-for-learning-to","slug":"meltr-meta-loss-transformer-for-learning-to","title":"MELTR: Meta Loss Transformer for Learning to Fine-tune Video Foundation Models","date":"2023-03-23","arxiv_id":"2303.13009","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/meltr-meta-loss-transformer-for-learning-to#ran","syntology_url":"https://syntology.ai/paper/2303.13009","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.13009"}},"official":{"repos":["mlvlab/MELTR"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tifa-accurate-and-interpretable-text-to-image","slug":"tifa-accurate-and-interpretable-text-to-image","title":"TIFA: Accurate and Interpretable Text-to-Image Faithfulness Evaluation with Question Answering","date":"2023-03-21","arxiv_id":"2303.11897","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tifa-accurate-and-interpretable-text-to-image#ran","syntology_url":"https://syntology.ai/paper/2303.11897","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.11897"}},"official":{"repos":["Yushi-Hu/tifa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ep-alm-efficient-perceptual-augmentation-of","slug":"ep-alm-efficient-perceptual-augmentation-of","title":"eP-ALM: Efficient Perceptual Augmentation of Language Models","date":"2023-03-20","arxiv_id":"2303.11403","repositories_listed":1,"syntology":{"n":16,"n_ran":15,"n_constructed":0,"n_ran_checked":10,"n_instrument":5,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":5,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ep-alm-efficient-perceptual-augmentation-of#ran","syntology_url":"https://syntology.ai/paper/2303.11403","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.11403"}},"official":{"repos":["mshukor/ep-alm"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/gpt-4-technical-report-1","slug":"gpt-4-technical-report-1","title":"GPT-4 Technical Report","date":"2023-03-15","arxiv_id":"2303.08774","repositories_listed":11,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":3,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 2 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/gpt-4-technical-report-1#ran","syntology_url":"https://syntology.ai/paper/2303.08774","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.08774"}},"official":{"repos":["openai/evals"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/pmc-clip-contrastive-language-image-pre","slug":"pmc-clip-contrastive-language-image-pre","title":"PMC-CLIP: Contrastive Language-Image Pre-training using Biomedical Documents","date":"2023-03-13","arxiv_id":"2303.07240","repositories_listed":2,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":5,"n_instrument":4,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 4 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/pmc-clip-contrastive-language-image-pre#ran","syntology_url":"https://syntology.ai/paper/2303.07240","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.07240"}},"official":{"repos":["WeixiongLin/PMC-CLIP"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/open-ended-medical-visual-question-answering","slug":"open-ended-medical-visual-question-answering","title":"Open-Ended Medical Visual Question Answering Through Prefix Tuning of Language Models","date":"2023-03-10","arxiv_id":"2303.05977","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/open-ended-medical-visual-question-answering#ran","syntology_url":"https://syntology.ai/paper/2303.05977","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.05977"}},"official":{"repos":["tjvsonsbeek/open-ended-medical-vqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/prompting-large-language-models-with-answer","slug":"prompting-large-language-models-with-answer","title":"Prophet: Prompting Large Language Models with Complementary Answer Heuristics for Knowledge-based Visual Question Answering","date":"2023-03-03","arxiv_id":"2303.01903","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/prompting-large-language-models-with-answer#ran","syntology_url":"https://syntology.ai/paper/2303.01903","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.01903"}},"official":{"repos":["milvlg/prophet"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/can-pre-trained-vision-and-language-models","slug":"can-pre-trained-vision-and-language-models","title":"Can Pre-trained Vision and Language Models Answer Visual Information-Seeking Questions?","date":"2023-02-23","arxiv_id":"2302.11713","repositories_listed":2,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/can-pre-trained-vision-and-language-models#ran","syntology_url":"https://syntology.ai/paper/2302.11713","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.11713"}},"official":{"repos":["edchengg/infoseek_eval"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/open-domain-visual-entity-recognition-towards","slug":"open-domain-visual-entity-recognition-towards","title":"Open-domain Visual Entity Recognition: Towards Recognizing Millions of Wikipedia Entities","date":"2023-02-22","arxiv_id":"2302.11154","repositories_listed":2,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/open-domain-visual-entity-recognition-towards#ran","syntology_url":"https://syntology.ai/paper/2302.11154","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.11154"}},"official":{"repos":["edchengg/oven_eval"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/uniadapter-unified-parameter-efficient","slug":"uniadapter-unified-parameter-efficient","title":"UniAdapter: Unified Parameter-Efficient Transfer Learning for Cross-modal Modeling","date":"2023-02-13","arxiv_id":"2302.06605","repositories_listed":2,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":4,"n_instrument":4,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/uniadapter-unified-parameter-efficient#ran","syntology_url":"https://syntology.ai/paper/2302.06605","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.06605"}},"official":{"repos":["rerv/uniadapter","uniadapter/uniadapter"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/mplug-2-a-modularized-multi-modal-foundation","slug":"mplug-2-a-modularized-multi-modal-foundation","title":"mPLUG-2: A Modularized Multi-modal Foundation Model Across Text, Image and Video","date":"2023-02-01","arxiv_id":"2302.00402","repositories_listed":4,"syntology":{"n":19,"n_ran":17,"n_constructed":0,"n_ran_checked":9,"n_instrument":8,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":2,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 8 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mplug-2-a-modularized-multi-modal-foundation#ran","syntology_url":"https://syntology.ai/paper/2302.00402","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.00402"}},"official":{"repos":["alibaba/AliceMind"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/blip-2-bootstrapping-language-image-pre","slug":"blip-2-bootstrapping-language-image-pre","title":"BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models","date":"2023-01-30","arxiv_id":"2301.12597","repositories_listed":17,"syntology":{"n":8,"n_ran":4,"n_constructed":4,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","sample_list":"/paper/blip-2-bootstrapping-language-image-pre#ran","syntology_url":"https://syntology.ai/paper/2301.12597","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.12597"}},"official":{"repos":["salesforce/lavis"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/harnessing-the-power-of-multi-task","slug":"harnessing-the-power-of-multi-task","title":"Harnessing the Power of Multi-Task Pretraining for Ground-Truth Level Natural Language Explanations","date":"2022-12-08","arxiv_id":"2212.04231","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/harnessing-the-power-of-multi-task#ran","syntology_url":"https://syntology.ai/paper/2212.04231","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.04231"}},"official":{"repos":["ofa-x/ofa-x"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/internvideo-general-video-foundation-models","slug":"internvideo-general-video-foundation-models","title":"InternVideo: General Video Foundation Models via Generative and Discriminative Learning","date":"2022-12-06","arxiv_id":"2212.03191","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/internvideo-general-video-foundation-models#ran","syntology_url":"https://syntology.ai/paper/2212.03191","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.03191"}},"official":{"repos":["opengvlab/internvideo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unifying-vision-text-and-layout-for-universal","slug":"unifying-vision-text-and-layout-for-universal","title":"Unifying Vision, Text, and Layout for Universal Document Processing","date":"2022-12-05","arxiv_id":"2212.02623","repositories_listed":5,"syntology":{"n":17,"n_ran":15,"n_constructed":0,"n_ran_checked":14,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":2,"n_no_contract":12,"n_pointer_only":4,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 2 violated, 12 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/unifying-vision-text-and-layout-for-universal#ran","syntology_url":"https://syntology.ai/paper/2212.02623","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.02623"}},"official":{"repos":["microsoft/i-code","microsoft/udop"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"url":"/paper/super-clevr-a-virtual-benchmark-to-diagnose","slug":"super-clevr-a-virtual-benchmark-to-diagnose","title":"Super-CLEVR: A Virtual Benchmark to Diagnose Domain Robustness in Visual Reasoning","date":"2022-12-01","arxiv_id":"2212.00259","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/super-clevr-a-virtual-benchmark-to-diagnose#ran","syntology_url":"https://syntology.ai/paper/2212.00259","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.00259"}},"official":{"repos":["lizw14/super-clevr"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/self-supervised-vision-language-pretraining","slug":"self-supervised-vision-language-pretraining","title":"Self-supervised vision-language pretraining for Medical visual question answering","date":"2022-11-24","arxiv_id":"2211.13594","repositories_listed":2,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":1,"n_instrument":5,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 5 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/self-supervised-vision-language-pretraining#ran","syntology_url":"https://syntology.ai/paper/2211.13594","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.13594"}},"official":{"repos":["pengfeiliheu/m2i2"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/x-2-vlm-all-in-one-pre-trained-model-for","slug":"x-2-vlm-all-in-one-pre-trained-model-for","title":"X$^2$-VLM: All-In-One Pre-trained Model For Vision-Language Tasks","date":"2022-11-22","arxiv_id":"2211.12402","repositories_listed":2,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":6,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/x-2-vlm-all-in-one-pre-trained-model-for#ran","syntology_url":"https://syntology.ai/paper/2211.12402","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.12402"}},"official":{"repos":["zengyan-97/x2-vlm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/expectation-maximization-contrastive-learning","slug":"expectation-maximization-contrastive-learning","title":"Expectation-Maximization Contrastive Learning for Compact Video-and-Language Representations","date":"2022-11-21","arxiv_id":"2211.11427","repositories_listed":4,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/expectation-maximization-contrastive-learning#ran","syntology_url":"https://syntology.ai/paper/2211.11427","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.11427"}},"official":{"repos":["jpthu17/emcl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/visual-programming-compositional-visual","slug":"visual-programming-compositional-visual","title":"Visual Programming: Compositional visual reasoning without training","date":"2022-11-18","arxiv_id":"2211.11559","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/visual-programming-compositional-visual#ran","syntology_url":"https://syntology.ai/paper/2211.11559","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.11559"}},"official":null}},{"url":"/paper/i-can-t-believe-there-s-no-images-learning","slug":"i-can-t-believe-there-s-no-images-learning","title":"I Can't Believe There's No Images! Learning Visual Tasks Using only Language Supervision","date":"2022-11-17","arxiv_id":"2211.09778","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/i-can-t-believe-there-s-no-images-learning#ran","syntology_url":"https://syntology.ai/paper/2211.09778","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.09778"}},"official":{"repos":["allenai/close"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/disentangling-aesthetic-and-technical-effects","slug":"disentangling-aesthetic-and-technical-effects","title":"Exploring Video Quality Assessment on User Generated Contents from Aesthetic and Technical Perspectives","date":"2022-11-09","arxiv_id":"2211.04894","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":3,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 3 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/disentangling-aesthetic-and-technical-effects#ran","syntology_url":"https://syntology.ai/paper/2211.04894","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.04894"}},"official":{"repos":["vqassessment/dover","QualityAssessment/DOVER"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cripp-vqa-counterfactual-reasoning-about","slug":"cripp-vqa-counterfactual-reasoning-about","title":"CRIPP-VQA: Counterfactual Reasoning about Implicit Physical Properties via Video Question Answering","date":"2022-11-07","arxiv_id":"2211.03779","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":2,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cripp-vqa-counterfactual-reasoning-about#ran","syntology_url":"https://syntology.ai/paper/2211.03779","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.03779"}},"official":{"repos":["maitreyapatel/cripp-vqa"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/vlc-bert-visual-question-answering-with","slug":"vlc-bert-visual-question-answering-with","title":"VLC-BERT: Visual Question Answering with Contextualized Commonsense Knowledge","date":"2022-10-24","arxiv_id":"2210.13626","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vlc-bert-visual-question-answering-with#ran","syntology_url":"https://syntology.ai/paper/2210.13626","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.13626"}},"official":{"repos":["aditya10/vlc-bert"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/plug-and-play-vqa-zero-shot-vqa-by-conjoining","slug":"plug-and-play-vqa-zero-shot-vqa-by-conjoining","title":"Plug-and-Play VQA: Zero-shot VQA by Conjoining Large Pretrained Models with Zero Training","date":"2022-10-17","arxiv_id":"2210.08773","repositories_listed":3,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/plug-and-play-vqa-zero-shot-vqa-by-conjoining#ran","syntology_url":"https://syntology.ai/paper/2210.08773","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.08773"}},"official":{"repos":["salesforce/lavis"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}}],"record_sha256":"1db904983b62eae1a7d2fbd8ee2dd3cf70dc6e72e2fb3ee86bb78310d17cac82","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}