{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/image-captioning/papers/ran/1","list_of":"/task/image-captioning","task":"Image Captioning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":3,"rows_per_page":100,"rows":[1,100],"of":243,"counts":{"archive_papers_tagged":1878,"with_a_code_link":774,"where_syntology_ran_a_sample":243,"not_listed_spam_title":0,"listed":1878,"listed_where_code_ran":243,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":201,"every_run_a_failure_of_syntologys_instrument":42,"listed_with_a_run_with_no_instrument_failure":201,"listed_every_run_a_failure_of_syntologys_instrument":42,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/image-captioning/papers/ran/1","prev":null,"next":"/task/image-captioning/papers/ran/2","papers":[{"url":"/paper/discovla-discrepancy-reduction-in-vision-1","slug":"discovla-discrepancy-reduction-in-vision-1","title":"DiscoVLA: Discrepancy Reduction in Vision, Language, and Alignment for Parameter-Efficient Video-Text Retrieval","date":"2025-06-10","arxiv_id":"2506.08887","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/discovla-discrepancy-reduction-in-vision-1#ran","syntology_url":"https://syntology.ai/paper/2506.08887","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.08887"}},"official":{"repos":["lunarshen/dsicovla"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/an-open-source-software-toolkit-benchmark","slug":"an-open-source-software-toolkit-benchmark","title":"An Open-Source Software Toolkit & Benchmark Suite for the Evaluation and Adaptation of Multimodal Action Models","date":"2025-06-10","arxiv_id":"2506.09172","repositories_listed":0,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/an-open-source-software-toolkit-benchmark#ran","syntology_url":"https://syntology.ai/paper/2506.09172","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.09172"}},"official":null}},{"url":"/paper/correlating-instruction-tuning-in-multimodal","slug":"correlating-instruction-tuning-in-multimodal","title":"Correlating instruction-tuning (in multimodal models) with vision-language processing (in the brain)","date":"2025-05-26","arxiv_id":"2505.20029","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/correlating-instruction-tuning-in-multimodal#ran","syntology_url":"https://syntology.ai/paper/2505.20029","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.20029"}},"official":{"repos":["subbareddy248/mllm_instruction_brain"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mind-the-gap-benchmarking-spatial-reasoning","slug":"mind-the-gap-benchmarking-spatial-reasoning","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","date":"2025-03-25","arxiv_id":"2503.19707","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mind-the-gap-benchmarking-spatial-reasoning#ran","syntology_url":"https://syntology.ai/paper/2503.19707","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.19707"}},"official":{"repos":["stogiannidis/srbench"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/weakly-supervised-video-scene-graph","slug":"weakly-supervised-video-scene-graph","title":"Weakly Supervised Video Scene Graph Generation via Natural Language Supervision","date":"2025-02-21","arxiv_id":"2502.15370","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/weakly-supervised-video-scene-graph#ran","syntology_url":"https://syntology.ai/paper/2502.15370","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.15370"}},"official":{"repos":["rlqja1107/NL-VSGG"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pretrained-image-text-models-are-secretly","slug":"pretrained-image-text-models-are-secretly","title":"Pretrained Image-Text Models are Secretly Video Captioners","date":"2025-02-19","arxiv_id":"2502.13363","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pretrained-image-text-models-are-secretly#ran","syntology_url":"https://syntology.ai/paper/2502.13363","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.13363"}},"official":{"repos":["chunhuizng/mllm-video-captioner"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/valley2-exploring-multimodal-models-with","slug":"valley2-exploring-multimodal-models-with","title":"Valley2: Exploring Multimodal Models with Scalable Vision-Language Design","date":"2025-01-10","arxiv_id":"2501.05901","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/valley2-exploring-multimodal-models-with#ran","syntology_url":"https://syntology.ai/paper/2501.05901","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.05901"}},"official":{"repos":["bytedance/valley"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/generalizing-from-simple-to-hard-visual","slug":"generalizing-from-simple-to-hard-visual","title":"Generalizing from SIMPLE to HARD Visual Reasoning: Can We Mitigate Modality Imbalance in VLMs?","date":"2025-01-05","arxiv_id":"2501.02669","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generalizing-from-simple-to-hard-visual#ran","syntology_url":"https://syntology.ai/paper/2501.02669","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.02669"}},"official":{"repos":["princeton-pli/vlm_s2h"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-high-quality-text-rich-image-instruction","slug":"a-high-quality-text-rich-image-instruction","title":"A High-Quality Text-Rich Image Instruction Tuning Dataset via Hybrid Instruction Generation","date":"2024-12-20","arxiv_id":"2412.16364","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-high-quality-text-rich-image-instruction#ran","syntology_url":"https://syntology.ai/paper/2412.16364","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.16364"}},"official":{"repos":["llavar/llavar-2"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/g-veval-a-versatile-metric-for-evaluating","slug":"g-veval-a-versatile-metric-for-evaluating","title":"G-VEval: A Versatile Metric for Evaluating Image and Video Captions Using GPT-4o","date":"2024-12-18","arxiv_id":"2412.13647","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/g-veval-a-versatile-metric-for-evaluating#ran","syntology_url":"https://syntology.ai/paper/2412.13647","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.13647"}},"official":{"repos":["ztangaj/gveval"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-large-vision-language-models-via","slug":"benchmarking-large-vision-language-models-via","title":"Benchmarking Large Vision-Language Models via Directed Scene Graph for Comprehensive Image Captioning","date":"2024-12-11","arxiv_id":"2412.08614","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-large-vision-language-models-via#ran","syntology_url":"https://syntology.ai/paper/2412.08614","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.08614"}},"official":{"repos":["lufan31/comprecap"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fg-cxr-a-radiologist-aligned-gaze-dataset-for","slug":"fg-cxr-a-radiologist-aligned-gaze-dataset-for","title":"FG-CXR: A Radiologist-Aligned Gaze Dataset for Enhancing Interpretability in Chest X-Ray Report Generation","date":"2024-11-23","arxiv_id":"2411.15413","repositories_listed":1,"syntology":{"n":5,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/fg-cxr-a-radiologist-aligned-gaze-dataset-for#ran","syntology_url":"https://syntology.ai/paper/2411.15413","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.15413"}},"official":{"repos":["uark-aicv/fg-cxr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/the-power-of-many-multi-agent-multimodal","slug":"the-power-of-many-multi-agent-multimodal","title":"The Power of Many: Multi-Agent Multimodal Models for Cultural Image Captioning","date":"2024-11-18","arxiv_id":"2411.11758","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-power-of-many-multi-agent-multimodal#ran","syntology_url":"https://syntology.ai/paper/2411.11758","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.11758"}},"official":{"repos":["michigannlp/mosaic"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/frontiers-in-intelligent-colonoscopy","slug":"frontiers-in-intelligent-colonoscopy","title":"Frontiers in Intelligent Colonoscopy","date":"2024-10-22","arxiv_id":"2410.17241","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/frontiers-in-intelligent-colonoscopy#ran","syntology_url":"https://syntology.ai/paper/2410.17241","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.17241"}},"official":{"repos":["ai4colonoscopy/intelliscope"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/tips-text-image-pretraining-with-spatial","slug":"tips-text-image-pretraining-with-spatial","title":"TIPS: Text-Image Pretraining with Spatial Awareness","date":"2024-10-21","arxiv_id":"2410.16512","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/tips-text-image-pretraining-with-spatial#ran","syntology_url":"https://syntology.ai/paper/2410.16512","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.16512"}},"official":null}},{"url":"/paper/remember-retrieve-and-generate-understanding","slug":"remember-retrieve-and-generate-understanding","title":"RAP: Retrieval-Augmented Personalization for Multimodal Large Language Models","date":"2024-10-17","arxiv_id":"2410.13360","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":10,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/remember-retrieve-and-generate-understanding#ran","syntology_url":"https://syntology.ai/paper/2410.13360","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.13360"}},"official":{"repos":["hoar012/rap-mllm"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/capeen-image-captioning-with-early-exits-and","slug":"capeen-image-captioning-with-early-exits-and","title":"CAPEEN: Image Captioning with Early Exits and Knowledge Distillation","date":"2024-10-06","arxiv_id":"2410.04433","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/capeen-image-captioning-with-early-exits-and#ran","syntology_url":"https://syntology.ai/paper/2410.04433","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.04433"}},"official":{"repos":["div290/capeen"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/molmo-and-pixmo-open-weights-and-open-data","slug":"molmo-and-pixmo-open-weights-and-open-data","title":"Molmo and PixMo: Open Weights and Open Data for State-of-the-Art Vision-Language Models","date":"2024-09-25","arxiv_id":"2409.17146","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/molmo-and-pixmo-open-weights-and-open-data#ran","syntology_url":"https://syntology.ai/paper/2409.17146","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.17146"}},"official":{"repos":["allenai/molmo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/journeybench-a-challenging-one-stop-vision","slug":"journeybench-a-challenging-one-stop-vision","title":"JourneyBench: A Challenging One-Stop Vision-Language Understanding Benchmark of Generated Images","date":"2024-09-19","arxiv_id":"2409.12953","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/journeybench-a-challenging-one-stop-vision#ran","syntology_url":"https://syntology.ai/paper/2409.12953","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.12953"}},"official":{"repos":["journeybench/journeybench"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/playground-v3-improving-text-to-image","slug":"playground-v3-improving-text-to-image","title":"Playground v3: Improving Text-to-Image Alignment with Deep-Fusion Large Language Models","date":"2024-09-16","arxiv_id":"2409.10695","repositories_listed":1,"syntology":{"n":21,"n_ran":18,"n_constructed":0,"n_ran_checked":9,"n_instrument":9,"n_unverified":3,"n_honours":4,"n_violates":1,"n_no_contract":4,"n_pointer_only":21,"phrase":"18 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 4 honoured, 1 violated, 4 with no contract checked; 9 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/playground-v3-improving-text-to-image#ran","syntology_url":"https://syntology.ai/paper/2409.10695","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.10695"}},"official":null}},{"url":"/paper/lime-m-less-is-more-for-evaluation-of-mllms","slug":"lime-m-less-is-more-for-evaluation-of-mllms","title":"LIME: Less Is More for MLLM Evaluation","date":"2024-09-10","arxiv_id":"2409.06851","repositories_listed":2,"syntology":{"n":14,"n_ran":13,"n_constructed":0,"n_ran_checked":7,"n_instrument":6,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":14,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 6 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lime-m-less-is-more-for-evaluation-of-mllms#ran","syntology_url":"https://syntology.ai/paper/2409.06851","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.06851"}},"official":{"repos":["kangreen0210/lime","kangreen0210/lime-m"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/training-free-zs-cir-via-weighted-modality","slug":"training-free-zs-cir-via-weighted-modality","title":"Training-free Zero-shot Composed Image Retrieval via Weighted Modality Fusion and Similarity","date":"2024-09-07","arxiv_id":"2409.04918","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/training-free-zs-cir-via-weighted-modality#ran","syntology_url":"https://syntology.ai/paper/2409.04918","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.04918"}},"official":{"repos":["whats2000/WeiMoCIR"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multimath-bridging-visual-and-mathematical","slug":"multimath-bridging-visual-and-mathematical","title":"MultiMath: Bridging Visual and Mathematical Reasoning for Large Language Models","date":"2024-08-30","arxiv_id":"2409.00147","repositories_listed":1,"syntology":{"n":18,"n_ran":17,"n_constructed":0,"n_ran_checked":9,"n_instrument":8,"n_unverified":1,"n_honours":0,"n_violates":3,"n_no_contract":6,"n_pointer_only":3,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 3 violated, 6 with no contract checked; 8 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multimath-bridging-visual-and-mathematical#ran","syntology_url":"https://syntology.ai/paper/2409.00147","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.00147"}},"official":{"repos":["pengshuai-rin/multimath"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/swift-semantic-watermarking-for-image-forgery","slug":"swift-semantic-watermarking-for-image-forgery","title":"SWIFT: Semantic Watermarking for Image Forgery Thwarting","date":"2024-07-26","arxiv_id":"2407.18995","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":1,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/swift-semantic-watermarking-for-image-forgery#ran","syntology_url":"https://syntology.ai/paper/2407.18995","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.18995"}},"official":{"repos":["gautierevn/swift_watermarking"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/vrsbench-a-versatile-vision-language","slug":"vrsbench-a-versatile-vision-language","title":"VRSBench: A Versatile Vision-Language Benchmark Dataset for Remote Sensing Image Understanding","date":"2024-06-18","arxiv_id":"2406.12384","repositories_listed":3,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":5,"n_instrument":6,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":13,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 6 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/vrsbench-a-versatile-vision-language#ran","syntology_url":"https://syntology.ai/paper/2406.12384","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12384"}},"official":{"repos":["lx709/vrsbench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/yo-llava-your-personalized-language-and","slug":"yo-llava-your-personalized-language-and","title":"Yo'LLaVA: Your Personalized Language and Vision Assistant","date":"2024-06-13","arxiv_id":"2406.09400","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/yo-llava-your-personalized-language-and#ran","syntology_url":"https://syntology.ai/paper/2406.09400","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09400"}},"official":{"repos":["WisconsinAIVision/YoLLaVA"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/imagenet3d-towards-general-purpose-object","slug":"imagenet3d-towards-general-purpose-object","title":"ImageNet3D: Towards General-Purpose Object-Level 3D Understanding","date":"2024-06-13","arxiv_id":"2406.09613","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/imagenet3d-towards-general-purpose-object#ran","syntology_url":"https://syntology.ai/paper/2406.09613","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09613"}},"official":{"repos":["wufeim/imagenet3d"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dragonfly-multi-resolution-zoom-supercharges","slug":"dragonfly-multi-resolution-zoom-supercharges","title":"Dragonfly: Multi-Resolution Zoom-In Encoding Enhances Vision-Language Models","date":"2024-06-03","arxiv_id":"2406.00977","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dragonfly-multi-resolution-zoom-supercharges#ran","syntology_url":"https://syntology.ai/paper/2406.00977","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.00977"}},"official":{"repos":["togethercomputer/dragonfly"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-and-improving-detail-image","slug":"benchmarking-and-improving-detail-image","title":"Benchmarking and Improving Detail Image Caption","date":"2024-05-29","arxiv_id":"2405.19092","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-and-improving-detail-image#ran","syntology_url":"https://syntology.ai/paper/2405.19092","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19092"}},"official":{"repos":["foundation-multimodal-models/capture"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rlaif-v-aligning-mllms-through-open-source-ai","slug":"rlaif-v-aligning-mllms-through-open-source-ai","title":"RLAIF-V: Open-Source AI Feedback Leads to Super GPT-4V Trustworthiness","date":"2024-05-27","arxiv_id":"2405.17220","repositories_listed":5,"syntology":{"n":22,"n_ran":17,"n_constructed":0,"n_ran_checked":10,"n_instrument":7,"n_unverified":5,"n_honours":0,"n_violates":2,"n_no_contract":8,"n_pointer_only":17,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 2 violated, 8 with no contract checked; 7 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/rlaif-v-aligning-mllms-through-open-source-ai#ran","syntology_url":"https://syntology.ai/paper/2405.17220","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.17220"}},"official":{"repos":["openbmb/omnilmm","rlhf-v/rlaif-v"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/unirag-universal-retrieval-augmentation-for","slug":"unirag-universal-retrieval-augmentation-for","title":"UniRAG: Universal Retrieval Augmentation for Large Vision Language Models","date":"2024-05-16","arxiv_id":"2405.10311","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":13,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/unirag-universal-retrieval-augmentation-for#ran","syntology_url":"https://syntology.ai/paper/2405.10311","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.10311"}},"official":{"repos":["castorini/unirag"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/cumo-scaling-multimodal-llm-with-co-upcycled","slug":"cumo-scaling-multimodal-llm-with-co-upcycled","title":"CuMo: Scaling Multimodal LLM with Co-Upcycled Mixture-of-Experts","date":"2024-05-09","arxiv_id":"2405.05949","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":6,"n_instrument":5,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cumo-scaling-multimodal-llm-with-co-upcycled#ran","syntology_url":"https://syntology.ai/paper/2405.05949","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.05949"}},"official":{"repos":["shi-labs/cumo"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/bridging-vision-and-language-spaces-with","slug":"bridging-vision-and-language-spaces-with","title":"Bridging Vision and Language Spaces with Assignment Prediction","date":"2024-04-15","arxiv_id":"2404.09632","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bridging-vision-and-language-spaces-with#ran","syntology_url":"https://syntology.ai/paper/2404.09632","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.09632"}},"official":{"repos":["park-jungin/vlap"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-visual-question-answering-through","slug":"enhancing-visual-question-answering-through","title":"Enhancing Visual Question Answering through Question-Driven Image Captions as Prompts","date":"2024-04-12","arxiv_id":"2404.08589","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/enhancing-visual-question-answering-through#ran","syntology_url":"https://syntology.ai/paper/2404.08589","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.08589"}},"official":{"repos":["ovguyo/captions-in-vqa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/view-selection-for-3d-captioning-via","slug":"view-selection-for-3d-captioning-via","title":"View Selection for 3D Captioning via Diffusion Ranking","date":"2024-04-11","arxiv_id":"2404.07984","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/view-selection-for-3d-captioning-via#ran","syntology_url":"https://syntology.ai/paper/2404.07984","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.07984"}},"official":null}},{"url":"/paper/comat-aligning-text-to-image-diffusion-model","slug":"comat-aligning-text-to-image-diffusion-model","title":"CoMat: Aligning Text-to-Image Diffusion Model with Image-to-Text Concept Matching","date":"2024-04-04","arxiv_id":"2404.03653","repositories_listed":2,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":9,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/comat-aligning-text-to-image-diffusion-model#ran","syntology_url":"https://syntology.ai/paper/2404.03653","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.03653"}},"official":{"repos":["caraj7/comat"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/harnessing-the-power-of-large-vision-language","slug":"harnessing-the-power-of-large-vision-language","title":"Harnessing the Power of Large Vision Language Models for Synthetic Image Detection","date":"2024-04-03","arxiv_id":"2404.02726","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/harnessing-the-power-of-large-vision-language#ran","syntology_url":"https://syntology.ai/paper/2404.02726","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.02726"}},"official":{"repos":["mamadou-keita/vlm-detect"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/disentangled-pre-training-for-human-object","slug":"disentangled-pre-training-for-human-object","title":"Disentangled Pre-training for Human-Object Interaction Detection","date":"2024-04-02","arxiv_id":"2404.01725","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/disentangled-pre-training-for-human-object#ran","syntology_url":"https://syntology.ai/paper/2404.01725","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.01725"}},"official":{"repos":["xingaoli/dp-hoi"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-by-correction-efficient-tuning-task","slug":"learning-by-correction-efficient-tuning-task","title":"Learning by Correction: Efficient Tuning Task for Zero-Shot Generative Vision-Language Reasoning","date":"2024-04-01","arxiv_id":"2404.00909","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":1,"n_ran_checked":4,"n_instrument":4,"n_unverified":2,"n_honours":1,"n_violates":2,"n_no_contract":1,"n_pointer_only":0,"phrase":"8 ran (of which 1 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 2 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/learning-by-correction-efficient-tuning-task#ran","syntology_url":"https://syntology.ai/paper/2404.00909","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.00909"}},"official":{"repos":["shtuplus/iccc_cvpr2024"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":1,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/h2rsvlm-towards-helpful-and-honest-remote","slug":"h2rsvlm-towards-helpful-and-honest-remote","title":"VHM: Versatile and Honest Vision Language Model for Remote Sensing Image Analysis","date":"2024-03-29","arxiv_id":"2403.20213","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/h2rsvlm-towards-helpful-and-honest-remote#ran","syntology_url":"https://syntology.ai/paper/2403.20213","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.20213"}},"official":{"repos":["opendatalab/h2rsvlm","opendatalab/vhm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/can-language-beat-numerical-regression","slug":"can-language-beat-numerical-regression","title":"Can Language Beat Numerical Regression? Language-Based Multimodal Trajectory Prediction","date":"2024-03-27","arxiv_id":"2403.18447","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":13,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/can-language-beat-numerical-regression#ran","syntology_url":"https://syntology.ai/paper/2403.18447","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.18447"}},"official":{"repos":["inhwanbae/lmtrajectory"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/cognitive-resilience-unraveling-the","slug":"cognitive-resilience-unraveling-the","title":"Cognitive resilience: Unraveling the proficiency of image-captioning models to interpret masked visual content","date":"2024-03-23","arxiv_id":"2403.15876","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cognitive-resilience-unraveling-the#ran","syntology_url":"https://syntology.ai/paper/2403.15876","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.15876"}},"official":{"repos":["dodoxxb/cognitive-resilience"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vl-icl-bench-the-devil-in-the-details-of","slug":"vl-icl-bench-the-devil-in-the-details-of","title":"VL-ICL Bench: The Devil in the Details of Multimodal In-Context Learning","date":"2024-03-19","arxiv_id":"2403.13164","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/vl-icl-bench-the-devil-in-the-details-of#ran","syntology_url":"https://syntology.ai/paper/2403.13164","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.13164"}},"official":{"repos":["ys-zong/vl-icl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-text-frozen-large-language-models-in","slug":"beyond-text-frozen-large-language-models-in","title":"Beyond Text: Frozen Large Language Models in Visual Signal Comprehension","date":"2024-03-12","arxiv_id":"2403.07874","repositories_listed":1,"syntology":{"n":26,"n_ran":17,"n_constructed":0,"n_ran_checked":6,"n_instrument":11,"n_unverified":9,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":26,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 11 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/beyond-text-frozen-large-language-models-in#ran","syntology_url":"https://syntology.ai/paper/2403.07874","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.07874"}},"official":{"repos":["zh460045050/v2l-tokenizer"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/pixart-s-weak-to-strong-training-of-diffusion","slug":"pixart-s-weak-to-strong-training-of-diffusion","title":"PixArt-Σ: Weak-to-Strong Training of Diffusion Transformer for 4K Text-to-Image Generation","date":"2024-03-07","arxiv_id":"2403.04692","repositories_listed":2,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/pixart-s-weak-to-strong-training-of-diffusion#ran","syntology_url":"https://syntology.ai/paper/2403.04692","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.04692"}},"official":{"repos":["PixArt-alpha/PixArt-sigma"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/meacap-memory-augmented-zero-shot-image","slug":"meacap-memory-augmented-zero-shot-image","title":"MeaCap: Memory-Augmented Zero-shot Image Captioning","date":"2024-03-06","arxiv_id":"2403.03715","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":1,"n_ran_checked":4,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":10,"phrase":"6 ran (of which 1 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/meacap-memory-augmented-zero-shot-image#ran","syntology_url":"https://syntology.ai/paper/2403.03715","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.03715"}},"official":{"repos":["joeyz0z/meacap"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":1,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/vtg-gpt-tuning-free-zero-shot-video-temporal-1","slug":"vtg-gpt-tuning-free-zero-shot-video-temporal-1","title":"VTG-GPT: Tuning-Free Zero-Shot Video Temporal Grounding with GPT","date":"2024-03-04","arxiv_id":"2403.02076","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":4,"n_instrument":5,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vtg-gpt-tuning-free-zero-shot-video-temporal-1#ran","syntology_url":"https://syntology.ai/paper/2403.02076","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.02076"}},"official":{"repos":["YoucanBaby/VTG-GPT"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/differentially-private-representation","slug":"differentially-private-representation","title":"Differentially Private Representation Learning via Image Captioning","date":"2024-03-04","arxiv_id":"2403.02506","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/differentially-private-representation#ran","syntology_url":"https://syntology.ai/paper/2403.02506","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.02506"}},"official":{"repos":["facebookresearch/dpcap"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/what-is-missing-in-multilingual-visual","slug":"what-is-missing-in-multilingual-visual","title":"What Is Missing in Multilingual Visual Reasoning and How to Fix It","date":"2024-03-03","arxiv_id":"2403.01404","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/what-is-missing-in-multilingual-visual#ran","syntology_url":"https://syntology.ai/paper/2403.01404","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.01404"}},"official":{"repos":["yueqis/multilingual_visual_reasoning"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/polos-multimodal-metric-learning-from-human","slug":"polos-multimodal-metric-learning-from-human","title":"Polos: Multimodal Metric Learning from Human Feedback for Image Captioning","date":"2024-02-28","arxiv_id":"2402.18091","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/polos-multimodal-metric-learning-from-human#ran","syntology_url":"https://syntology.ai/paper/2402.18091","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.18091"}},"official":{"repos":["keio-smilab24/Polos"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/fine-tuning-clip-text-encoders-with-two-step","slug":"fine-tuning-clip-text-encoders-with-two-step","title":"Fine-tuning CLIP Text Encoders with Two-step Paraphrasing","date":"2024-02-23","arxiv_id":"2402.15120","repositories_listed":0,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fine-tuning-clip-text-encoders-with-two-step#ran","syntology_url":"https://syntology.ai/paper/2402.15120","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.15120"}},"official":null}},{"url":"/paper/chatearthnet-a-global-scale-high-quality","slug":"chatearthnet-a-global-scale-high-quality","title":"ChatEarthNet: A Global-Scale Image-Text Dataset Empowering Vision-Language Geo-Foundation Models","date":"2024-02-17","arxiv_id":"2402.11325","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/chatearthnet-a-global-scale-high-quality#ran","syntology_url":"https://syntology.ai/paper/2402.11325","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11325"}},"official":{"repos":["zhu-xlab/ChatEarthNet"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/gpts-are-multilingual-annotators-for-sequence","slug":"gpts-are-multilingual-annotators-for-sequence","title":"GPTs Are Multilingual Annotators for Sequence Generation Tasks","date":"2024-02-08","arxiv_id":"2402.05512","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/gpts-are-multilingual-annotators-for-sequence#ran","syntology_url":"https://syntology.ai/paper/2402.05512","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05512"}},"official":{"repos":["c-juhwan/gpt-multilingual-annotator"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/good-at-captioning-bad-at-counting","slug":"good-at-captioning-bad-at-counting","title":"Good at captioning, bad at counting: Benchmarking GPT-4V on Earth observation data","date":"2024-01-31","arxiv_id":"2401.17600","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/good-at-captioning-bad-at-counting#ran","syntology_url":"https://syntology.ai/paper/2401.17600","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.17600"}},"official":{"repos":["Earth-Intelligence-Lab/vleo-bench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/scimmir-benchmarking-scientific-multi-modal","slug":"scimmir-benchmarking-scientific-multi-modal","title":"SciMMIR: Benchmarking Scientific Multi-modal Information Retrieval","date":"2024-01-24","arxiv_id":"2401.13478","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":10,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/scimmir-benchmarking-scientific-multi-modal#ran","syntology_url":"https://syntology.ai/paper/2401.13478","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.13478"}},"official":{"repos":["wusiwei0410/scimmir"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mining-fine-grained-image-text-alignment-for","slug":"mining-fine-grained-image-text-alignment-for","title":"Mining Fine-Grained Image-Text Alignment for Zero-Shot Captioning via Text-Only Training","date":"2024-01-04","arxiv_id":"2401.02347","repositories_listed":1,"syntology":{"n":8,"n_ran":3,"n_constructed":1,"n_ran_checked":1,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":8,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/mining-fine-grained-image-text-alignment-for#ran","syntology_url":"https://syntology.ai/paper/2401.02347","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.02347"}},"official":{"repos":["artanic30/maccap"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":5,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/gpt-4v-ision-is-a-generalist-web-agent-if","slug":"gpt-4v-ision-is-a-generalist-web-agent-if","title":"GPT-4V(ision) is a Generalist Web Agent, if Grounded","date":"2024-01-03","arxiv_id":"2401.01614","repositories_listed":1,"syntology":{"n":14,"n_ran":13,"n_constructed":0,"n_ran_checked":12,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":14,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/gpt-4v-ision-is-a-generalist-web-agent-if#ran","syntology_url":"https://syntology.ai/paper/2401.01614","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.01614"}},"official":{"repos":["osu-nlp-group/seeact"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/tinygpt-v-efficient-multimodal-large-language","slug":"tinygpt-v-efficient-multimodal-large-language","title":"TinyGPT-V: Efficient Multimodal Large Language Model via Small Backbones","date":"2023-12-28","arxiv_id":"2312.16862","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":1,"n_instrument":6,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tinygpt-v-efficient-multimodal-large-language#ran","syntology_url":"https://syntology.ai/paper/2312.16862","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.16862"}},"official":{"repos":["dlyuangod/tinygpt-v"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-vision-from-models-rivals-learning","slug":"learning-vision-from-models-rivals-learning","title":"Learning Vision from Models Rivals Learning Vision from Data","date":"2023-12-28","arxiv_id":"2312.17742","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-vision-from-models-rivals-learning#ran","syntology_url":"https://syntology.ai/paper/2312.17742","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.17742"}},"official":{"repos":["google-research/syn-rep-learn"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vcoder-versatile-vision-encoders-for","slug":"vcoder-versatile-vision-encoders-for","title":"VCoder: Versatile Vision Encoders for Multimodal Large Language Models","date":"2023-12-21","arxiv_id":"2312.14233","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vcoder-versatile-vision-encoders-for#ran","syntology_url":"https://syntology.ai/paper/2312.14233","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.14233"}},"official":{"repos":["shi-labs/vcoder"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/icd-lm-configuring-vision-language-in-context","slug":"icd-lm-configuring-vision-language-in-context","title":"Lever LM: Configuring In-Context Sequence to Lever Large Vision Language Models","date":"2023-12-15","arxiv_id":"2312.10104","repositories_listed":2,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/icd-lm-configuring-vision-language-in-context#ran","syntology_url":"https://syntology.ai/paper/2312.10104","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.10104"}},"official":{"repos":["forjadeforest/icd-lm","forjadeforest/lever-lm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/a-picture-is-worth-more-than-77-text-tokens","slug":"a-picture-is-worth-more-than-77-text-tokens","title":"A Picture is Worth More Than 77 Text Tokens: Evaluating CLIP-Style Models on Dense Captions","date":"2023-12-14","arxiv_id":"2312.08578","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":12,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/a-picture-is-worth-more-than-77-text-tokens#ran","syntology_url":"https://syntology.ai/paper/2312.08578","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.08578"}},"official":{"repos":["facebookresearch/dci"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/genixer-empowering-multimodal-large-language","slug":"genixer-empowering-multimodal-large-language","title":"Genixer: Empowering Multimodal Large Language Models as a Powerful Data Generator","date":"2023-12-11","arxiv_id":"2312.06731","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":4,"n_pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 1 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/genixer-empowering-multimodal-large-language#ran","syntology_url":"https://syntology.ai/paper/2312.06731","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.06731"}},"official":{"repos":["zhaohengyuan1/genixer"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/quilt-llava-visual-instruction-tuning-by","slug":"quilt-llava-visual-instruction-tuning-by","title":"Quilt-LLaVA: Visual Instruction Tuning by Extracting Localized Narratives from Open-Source Histopathology Videos","date":"2023-12-07","arxiv_id":"2312.04746","repositories_listed":2,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/quilt-llava-visual-instruction-tuning-by#ran","syntology_url":"https://syntology.ai/paper/2312.04746","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.04746"}},"official":null}},{"url":"/paper/mocha-multi-objective-reinforcement","slug":"mocha-multi-objective-reinforcement","title":"Mitigating Open-Vocabulary Caption Hallucinations","date":"2023-12-06","arxiv_id":"2312.03631","repositories_listed":1,"syntology":{"n":15,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":15,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/mocha-multi-objective-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2312.03631","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.03631"}},"official":{"repos":["assafbk/mocha_code"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/llama-vid-an-image-is-worth-2-tokens-in-large","slug":"llama-vid-an-image-is-worth-2-tokens-in-large","title":"LLaMA-VID: An Image is Worth 2 Tokens in Large Language Models","date":"2023-11-28","arxiv_id":"2311.17043","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/llama-vid-an-image-is-worth-2-tokens-in-large#ran","syntology_url":"https://syntology.ai/paper/2311.17043","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.17043"}},"official":{"repos":["dvlab-research/llama-vid"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mobileclip-fast-image-text-models-through","slug":"mobileclip-fast-image-text-models-through","title":"MobileCLIP: Fast Image-Text Models through Multi-Modal Reinforced Training","date":"2023-11-28","arxiv_id":"2311.17049","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mobileclip-fast-image-text-models-through#ran","syntology_url":"https://syntology.ai/paper/2311.17049","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.17049"}},"official":{"repos":["apple/ml-mobileclip","rwightman/pytorch-image-models"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/zero-shot-audio-captioning-with-audio","slug":"zero-shot-audio-captioning-with-audio","title":"Zero-shot audio captioning with audio-language model guidance and audio context keywords","date":"2023-11-14","arxiv_id":"2311.08396","repositories_listed":1,"syntology":{"n":17,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":17,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/zero-shot-audio-captioning-with-audio#ran","syntology_url":"https://syntology.ai/paper/2311.08396","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.08396"}},"official":{"repos":["explainableml/zeraucap"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/monkey-image-resolution-and-text-label-are","slug":"monkey-image-resolution-and-text-label-are","title":"Monkey: Image Resolution and Text Label Are Important Things for Large Multi-modal Models","date":"2023-11-11","arxiv_id":"2311.06607","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/monkey-image-resolution-and-text-label-are#ran","syntology_url":"https://syntology.ai/paper/2311.06607","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.06607"}},"official":{"repos":["yuliang-liu/monkey"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/glamm-pixel-grounding-large-multimodal-model","slug":"glamm-pixel-grounding-large-multimodal-model","title":"GLaMM: Pixel Grounding Large Multimodal Model","date":"2023-11-06","arxiv_id":"2311.03356","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/glamm-pixel-grounding-large-multimodal-model#ran","syntology_url":"https://syntology.ai/paper/2311.03356","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.03356"}},"official":{"repos":["mbzuai-oryx/groundingLMM"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/language-guided-visual-question-answering","slug":"language-guided-visual-question-answering","title":"Language Guided Visual Question Answering: Elevate Your Multimodal Language Model Using Knowledge-Enriched Prompts","date":"2023-10-31","arxiv_id":"2310.20159","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/language-guided-visual-question-answering#ran","syntology_url":"https://syntology.ai/paper/2310.20159","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.20159"}},"official":{"repos":["declare-lab/lg-vqa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/myriad-large-multimodal-model-by-applying","slug":"myriad-large-multimodal-model-by-applying","title":"Myriad: Large Multimodal Model by Applying Vision Experts for Industrial Anomaly Detection","date":"2023-10-29","arxiv_id":"2310.19070","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":4,"n_honours":1,"n_violates":2,"n_no_contract":2,"n_pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 2 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/myriad-large-multimodal-model-by-applying#ran","syntology_url":"https://syntology.ai/paper/2310.19070","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.19070"}},"official":{"repos":["tzjtatata/myriad"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-grounding-helps-learn-word-meanings-in","slug":"visual-grounding-helps-learn-word-meanings-in","title":"Visual Grounding Helps Learn Word Meanings in Low-Data Regimes","date":"2023-10-20","arxiv_id":"2310.13257","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/visual-grounding-helps-learn-word-meanings-in#ran","syntology_url":"https://syntology.ai/paper/2310.13257","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.13257"}},"official":{"repos":["EvLab-MIT/LexiContrastiveGrd"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/language-models-as-knowledge-bases-for-visual","slug":"language-models-as-knowledge-bases-for-visual","title":"Language Models as Knowledge Bases for Visual Word Sense Disambiguation","date":"2023-10-03","arxiv_id":"2310.01960","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/language-models-as-knowledge-bases-for-visual#ran","syntology_url":"https://syntology.ai/paper/2310.01960","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.01960"}},"official":{"repos":["anastasiakrith/llm-for-vwsd"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/sieve-multimodal-dataset-pruning-using-image","slug":"sieve-multimodal-dataset-pruning-using-image","title":"Sieve: Multimodal Dataset Pruning Using Image Captioning Models","date":"2023-10-03","arxiv_id":"2310.02110","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sieve-multimodal-dataset-pruning-using-image#ran","syntology_url":"https://syntology.ai/paper/2310.02110","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.02110"}},"official":{"repos":["facebookresearch/sieve"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mathvista-evaluating-mathematical-reasoning","slug":"mathvista-evaluating-mathematical-reasoning","title":"MathVista: Evaluating Mathematical Reasoning of Foundation Models in Visual Contexts","date":"2023-10-03","arxiv_id":"2310.02255","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mathvista-evaluating-mathematical-reasoning#ran","syntology_url":"https://syntology.ai/paper/2310.02255","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.02255"}},"official":null}},{"url":"/paper/beyond-generation-harnessing-text-to-image","slug":"beyond-generation-harnessing-text-to-image","title":"Beyond Generation: Harnessing Text to Image Models for Object Detection and Segmentation","date":"2023-09-12","arxiv_id":"2309.05956","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/beyond-generation-harnessing-text-to-image#ran","syntology_url":"https://syntology.ai/paper/2309.05956","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.05956"}},"official":{"repos":["gyhandy/text2image-for-detection"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/exchanging-based-multimodal-fusion-with","slug":"exchanging-based-multimodal-fusion-with","title":"Exchanging-based Multimodal Fusion with Transformer","date":"2023-09-05","arxiv_id":"2309.02190","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/exchanging-based-multimodal-fusion-with#ran","syntology_url":"https://syntology.ai/paper/2309.02190","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.02190"}},"official":{"repos":["recklessronan/muse"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cliptrans-transferring-visual-knowledge-with","slug":"cliptrans-transferring-visual-knowledge-with","title":"CLIPTrans: Transferring Visual Knowledge with Pre-trained Models for Multimodal Machine Translation","date":"2023-08-29","arxiv_id":"2308.15226","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/cliptrans-transferring-visual-knowledge-with#ran","syntology_url":"https://syntology.ai/paper/2308.15226","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.15226"}},"official":{"repos":["devaansh100/cliptrans"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/multicapclip-auto-encoding-prompts-for-zero","slug":"multicapclip-auto-encoding-prompts-for-zero","title":"MultiCapCLIP: Auto-Encoding Prompts for Zero-Shot Multilingual Visual Captioning","date":"2023-08-25","arxiv_id":"2308.13218","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/multicapclip-auto-encoding-prompts-for-zero#ran","syntology_url":"https://syntology.ai/paper/2308.13218","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.13218"}},"official":{"repos":["yangbang18/multicapclip"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vigc-visual-instruction-generation-and","slug":"vigc-visual-instruction-generation-and","title":"VIGC: Visual Instruction Generation and Correction","date":"2023-08-24","arxiv_id":"2308.12714","repositories_listed":2,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":1,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/vigc-visual-instruction-generation-and#ran","syntology_url":"https://syntology.ai/paper/2308.12714","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.12714"}},"official":{"repos":["opendatalab/vigc"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/with-a-little-help-from-your-own-past","slug":"with-a-little-help-from-your-own-past","title":"With a Little Help from your own Past: Prototypical Memory Networks for Image Captioning","date":"2023-08-23","arxiv_id":"2308.12383","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/with-a-little-help-from-your-own-past#ran","syntology_url":"https://syntology.ai/paper/2308.12383","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.12383"}},"official":{"repos":["aimagelab/pma-net"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vl-pet-vision-and-language-parameter","slug":"vl-pet-vision-and-language-parameter","title":"VL-PET: Vision-and-Language Parameter-Efficient Tuning via Granularity Control","date":"2023-08-18","arxiv_id":"2308.09804","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vl-pet-vision-and-language-parameter#ran","syntology_url":"https://syntology.ai/paper/2308.09804","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09804"}},"official":{"repos":["henryhzy/vl-pet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pro-cap-leveraging-a-frozen-vision-language","slug":"pro-cap-leveraging-a-frozen-vision-language","title":"Pro-Cap: Leveraging a Frozen Vision-Language Model for Hateful Meme Detection","date":"2023-08-16","arxiv_id":"2308.08088","repositories_listed":2,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/pro-cap-leveraging-a-frozen-vision-language#ran","syntology_url":"https://syntology.ai/paper/2308.08088","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.08088"}},"official":{"repos":["social-ai-studio/pro-cap"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/git-mol-a-multi-modal-large-language-model","slug":"git-mol-a-multi-modal-large-language-model","title":"GIT-Mol: A Multi-modal Large Language Model for Molecular Science with Graph, Image, and Text","date":"2023-08-14","arxiv_id":"2308.06911","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/git-mol-a-multi-modal-large-language-model#ran","syntology_url":"https://syntology.ai/paper/2308.06911","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.06911"}},"official":{"repos":["ai-hpc-research-team/git-mol"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/empowering-vision-language-models-to-follow","slug":"empowering-vision-language-models-to-follow","title":"Fine-tuning Multimodal LLMs to Follow Zero-shot Demonstrative Instructions","date":"2023-08-08","arxiv_id":"2308.04152","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":2,"n_instrument":8,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 8 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/empowering-vision-language-models-to-follow#ran","syntology_url":"https://syntology.ai/paper/2308.04152","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.04152"}},"official":{"repos":["dcdmllm/cheetah"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/transferable-decoding-with-visual-entities","slug":"transferable-decoding-with-visual-entities","title":"Transferable Decoding with Visual Entities for Zero-Shot Image Captioning","date":"2023-07-31","arxiv_id":"2307.16525","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":3,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"6 ran (of which 3 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/transferable-decoding-with-visual-entities#ran","syntology_url":"https://syntology.ai/paper/2307.16525","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.16525"}},"official":{"repos":["feielysia/viecap"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":3,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mblip-efficient-bootstrapping-of-multilingual","slug":"mblip-efficient-bootstrapping-of-multilingual","title":"mBLIP: Efficient Bootstrapping of Multilingual Vision-LLMs","date":"2023-07-13","arxiv_id":"2307.06930","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mblip-efficient-bootstrapping-of-multilingual#ran","syntology_url":"https://syntology.ai/paper/2307.06930","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.06930"}},"official":{"repos":["gregor-ge/mblip"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/llavar-enhanced-visual-instruction-tuning-for","slug":"llavar-enhanced-visual-instruction-tuning-for","title":"LLaVAR: Enhanced Visual Instruction Tuning for Text-Rich Image Understanding","date":"2023-06-29","arxiv_id":"2306.17107","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":0,"n_instrument":6,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/llavar-enhanced-visual-instruction-tuning-for#ran","syntology_url":"https://syntology.ai/paper/2306.17107","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.17107"}},"official":{"repos":["SALT-NLP/LLaVAR"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official","unlocated"]}}},{"url":"/paper/palm-predicting-actions-through-language","slug":"palm-predicting-actions-through-language","title":"Palm: Predicting Actions through Language Models @ Ego4D Long-Term Action Anticipation Challenge 2023","date":"2023-06-28","arxiv_id":"2306.16545","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/palm-predicting-actions-through-language#ran","syntology_url":"https://syntology.ai/paper/2306.16545","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.16545"}},"official":{"repos":["dandoge/palm"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/shikra-unleashing-multimodal-llm-s","slug":"shikra-unleashing-multimodal-llm-s","title":"Shikra: Unleashing Multimodal LLM's Referential Dialogue Magic","date":"2023-06-27","arxiv_id":"2306.15195","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/shikra-unleashing-multimodal-llm-s#ran","syntology_url":"https://syntology.ai/paper/2306.15195","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.15195"}},"official":{"repos":["shikras/shikra"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/investigating-prompting-techniques-for-zero","slug":"investigating-prompting-techniques-for-zero","title":"Investigating Prompting Techniques for Zero- and Few-Shot Visual Question Answering","date":"2023-06-16","arxiv_id":"2306.09996","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/investigating-prompting-techniques-for-zero#ran","syntology_url":"https://syntology.ai/paper/2306.09996","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.09996"}},"official":{"repos":["rabiulcste/vqazero"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/image-captioners-are-scalable-vision-learners","slug":"image-captioners-are-scalable-vision-learners","title":"Image Captioners Are Scalable Vision Learners Too","date":"2023-06-13","arxiv_id":"2306.07915","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/image-captioners-are-scalable-vision-learners#ran","syntology_url":"https://syntology.ai/paper/2306.07915","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.07915"}},"official":{"repos":["google-research/big_vision"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/understanding-and-mitigating-copying-in-1","slug":"understanding-and-mitigating-copying-in-1","title":"Understanding and Mitigating Copying in Diffusion Models","date":"2023-05-31","arxiv_id":"2305.20086","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/understanding-and-mitigating-copying-in-1#ran","syntology_url":"https://syntology.ai/paper/2305.20086","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.20086"}},"official":{"repos":["somepago/dcr"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/test-time-adaptation-with-clip-reward-for","slug":"test-time-adaptation-with-clip-reward-for","title":"Test-Time Adaptation with CLIP Reward for Zero-Shot Generalization in Vision-Language Models","date":"2023-05-29","arxiv_id":"2305.18010","repositories_listed":1,"syntology":{"n":10,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/test-time-adaptation-with-clip-reward-for#ran","syntology_url":"https://syntology.ai/paper/2305.18010","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.18010"}},"official":{"repos":["mzhaoshuai/rlcf"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/contextual-object-detection-with-multimodal","slug":"contextual-object-detection-with-multimodal","title":"Contextual Object Detection with Multimodal Large Language Models","date":"2023-05-29","arxiv_id":"2305.18279","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/contextual-object-detection-with-multimodal#ran","syntology_url":"https://syntology.ai/paper/2305.18279","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.18279"}},"official":{"repos":["yuhangzang/contextdet"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vast-a-vision-audio-subtitle-text-omni-1","slug":"vast-a-vision-audio-subtitle-text-omni-1","title":"VAST: A Vision-Audio-Subtitle-Text Omni-Modality Foundation Model and Dataset","date":"2023-05-29","arxiv_id":"2305.18500","repositories_listed":2,"syntology":{"n":42,"n_ran":35,"n_constructed":4,"n_ran_checked":29,"n_instrument":6,"n_unverified":7,"n_honours":2,"n_violates":1,"n_no_contract":26,"n_pointer_only":8,"phrase":"35 ran (of which 4 constructed an object rather than computing a result; 29 with no instrument failure: 2 honoured, 1 violated, 26 with no contract checked; 6 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/vast-a-vision-audio-subtitle-text-omni-1#ran","syntology_url":"https://syntology.ai/paper/2305.18500","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.18500"}},"official":{"repos":["txh-mercury/vast"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":4,"n_ran_no_instrument_failure":12,"n_unverified":7,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/crossget-cross-guided-ensemble-of-tokens-for","slug":"crossget-cross-guided-ensemble-of-tokens-for","title":"CrossGET: Cross-Guided Ensemble of Tokens for Accelerating Vision-Language Transformers","date":"2023-05-27","arxiv_id":"2305.17455","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/crossget-cross-guided-ensemble-of-tokens-for#ran","syntology_url":"https://syntology.ai/paper/2305.17455","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.17455"}},"official":{"repos":["sdc17/crossget"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/biomedgpt-a-unified-and-generalist-biomedical","slug":"biomedgpt-a-unified-and-generalist-biomedical","title":"BiomedGPT: A Generalist Vision-Language Foundation Model for Diverse Biomedical Tasks","date":"2023-05-26","arxiv_id":"2305.17100","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/biomedgpt-a-unified-and-generalist-biomedical#ran","syntology_url":"https://syntology.ai/paper/2305.17100","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.17100"}},"official":{"repos":["taokz/biomedgpt"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/exploring-diverse-in-context-configurations","slug":"exploring-diverse-in-context-configurations","title":"Exploring Diverse In-Context Configurations for Image Captioning","date":"2023-05-24","arxiv_id":"2305.14800","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/exploring-diverse-in-context-configurations#ran","syntology_url":"https://syntology.ai/paper/2305.14800","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14800"}},"official":{"repos":["yongliang-wu/explorecfg"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}}],"record_sha256":"5400329567d5f8f0d83716b8e7f17b9a439fe55adeb9041d5f7aa3189d5906dd","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}