{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/image-captioning/papers/3","list_of":"/task/image-captioning","task":"Image Captioning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":19,"rows_per_page":100,"rows":[201,300],"of":1878,"counts":{"archive_papers_tagged":1878,"with_a_code_link":774,"where_syntology_ran_a_sample":243,"not_listed_spam_title":0,"listed":1878,"listed_where_code_ran":243,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":201,"every_run_a_failure_of_syntologys_instrument":42,"listed_with_a_run_with_no_instrument_failure":201,"listed_every_run_a_failure_of_syntologys_instrument":42,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/image-captioning","prev":"/task/image-captioning/papers/2","next":"/task/image-captioning/papers/4","papers":[{"url":"/paper/generalizing-from-simple-to-hard-visual","slug":"generalizing-from-simple-to-hard-visual","title":"Generalizing from SIMPLE to HARD Visual Reasoning: Can We Mitigate Modality Imbalance in VLMs?","date":"2025-01-05","arxiv_id":"2501.02669","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generalizing-from-simple-to-hard-visual#ran","syntology_url":"https://syntology.ai/paper/2501.02669","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.02669"}},"official":{"repos":["princeton-pli/vlm_s2h"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/diffusion-bridge-leveraging-diffusion-model","slug":"diffusion-bridge-leveraging-diffusion-model","title":"Diffusion Bridge: Leveraging Diffusion Model to Reduce the Modality Gap Between Text and Vision for Zero-Shot Image Captioning","date":"2025-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/vipcap-retrieval-text-based-visual-prompts","slug":"vipcap-retrieval-text-based-visual-prompts","title":"ViPCap: Retrieval Text-Based Visual Prompts for Lightweight Image Captioning","date":"2024-12-26","arxiv_id":"2412.19289","repositories_listed":1,"syntology":null},{"url":"/paper/evalmuse-40k-a-reliable-and-fine-grained","slug":"evalmuse-40k-a-reliable-and-fine-grained","title":"EvalMuse-40K: A Reliable and Fine-Grained Benchmark with Comprehensive Human Annotations for Text-to-Image Generation Model Evaluation","date":"2024-12-24","arxiv_id":"2412.18150","repositories_listed":1,"syntology":null},{"url":"/paper/silvar-speech-driven-multimodal-model-for","slug":"silvar-speech-driven-multimodal-model-for","title":"SilVar: Speech Driven Multimodal Model for Reasoning Visual Question Answering and Object Localization","date":"2024-12-21","arxiv_id":"2412.16771","repositories_listed":1,"syntology":null},{"url":"/paper/a-high-quality-text-rich-image-instruction","slug":"a-high-quality-text-rich-image-instruction","title":"A High-Quality Text-Rich Image Instruction Tuning Dataset via Hybrid Instruction Generation","date":"2024-12-20","arxiv_id":"2412.16364","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-high-quality-text-rich-image-instruction#ran","syntology_url":"https://syntology.ai/paper/2412.16364","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.16364"}},"official":{"repos":["llavar/llavar-2"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-human-data-aligning-multimodal-large","slug":"beyond-human-data-aligning-multimodal-large","title":"Beyond Human Data: Aligning Multimodal Large Language Models by Iterative Self-Evolution","date":"2024-12-20","arxiv_id":"2412.15650","repositories_listed":1,"syntology":null},{"url":"/paper/reframing-image-difference-captioning-with","slug":"reframing-image-difference-captioning-with","title":"Reframing Image Difference Captioning with BLIP2IDC and Synthetic Augmentation","date":"2024-12-20","arxiv_id":"2412.15939","repositories_listed":1,"syntology":null},{"url":"/paper/unveiling-uncertainty-a-deep-dive-into","slug":"unveiling-uncertainty-a-deep-dive-into","title":"Unveiling Uncertainty: A Deep Dive into Calibration and Performance of Multimodal Large Language Models","date":"2024-12-19","arxiv_id":"2412.14660","repositories_listed":1,"syntology":null},{"url":"/paper/descriptive-caption-enhancement-with-visual","slug":"descriptive-caption-enhancement-with-visual","title":"Descriptive Caption Enhancement with Visual Specialists for Multimodal Perception","date":"2024-12-18","arxiv_id":"2412.14233","repositories_listed":1,"syntology":null},{"url":"/paper/g-veval-a-versatile-metric-for-evaluating","slug":"g-veval-a-versatile-metric-for-evaluating","title":"G-VEval: A Versatile Metric for Evaluating Image and Video Captions Using GPT-4o","date":"2024-12-18","arxiv_id":"2412.13647","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/g-veval-a-versatile-metric-for-evaluating#ran","syntology_url":"https://syntology.ai/paper/2412.13647","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.13647"}},"official":{"repos":["ztangaj/gveval"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/jovale-detecting-human-actions-in-video-using","slug":"jovale-detecting-human-actions-in-video-using","title":"JoVALE: Detecting Human Actions in Video Using Audiovisual and Language Contexts","date":"2024-12-18","arxiv_id":"2412.13708","repositories_listed":1,"syntology":null},{"url":"/paper/typhoon-2-a-family-of-open-text-and","slug":"typhoon-2-a-family-of-open-text-and","title":"Typhoon 2: A Family of Open Text and Multimodal Thai Large Language Models","date":"2024-12-18","arxiv_id":"2412.13702","repositories_listed":1,"syntology":null},{"url":"/paper/medmax-mixed-modal-instruction-tuning-for","slug":"medmax-mixed-modal-instruction-tuning-for","title":"MedMax: Mixed-Modal Instruction Tuning for Training Biomedical Assistants","date":"2024-12-17","arxiv_id":"2412.12661","repositories_listed":1,"syntology":null},{"url":"/paper/from-simple-to-professional-a-combinatorial","slug":"from-simple-to-professional-a-combinatorial","title":"From Simple to Professional: A Combinatorial Controllable Image Captioning Agent","date":"2024-12-15","arxiv_id":"2412.11025","repositories_listed":1,"syntology":null},{"url":"/paper/automated-image-captioning-with-cnns-and","slug":"automated-image-captioning-with-cnns-and","title":"Automated Image Captioning with CNNs and Transformers","date":"2024-12-13","arxiv_id":"2412.10511","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-large-vision-language-models-via","slug":"benchmarking-large-vision-language-models-via","title":"Benchmarking Large Vision-Language Models via Directed Scene Graph for Comprehensive Image Captioning","date":"2024-12-11","arxiv_id":"2412.08614","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-large-vision-language-models-via#ran","syntology_url":"https://syntology.ai/paper/2412.08614","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.08614"}},"official":{"repos":["lufan31/comprecap"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/automated-medical-report-generation-for-ecg","slug":"automated-medical-report-generation-for-ecg","title":"Automated Medical Report Generation for ECG Data: Bridging Medical Text and Signal Processing with Deep Learning","date":"2024-12-05","arxiv_id":"2412.04067","repositories_listed":1,"syntology":null},{"url":"/paper/florence-vl-enhancing-vision-language-models","slug":"florence-vl-enhancing-vision-language-models","title":"Florence-VL: Enhancing Vision-Language Models with Generative Vision Encoder and Depth-Breadth Fusion","date":"2024-12-05","arxiv_id":"2412.04424","repositories_listed":1,"syntology":null},{"url":"/paper/lab-rag-label-boosted-retrieval-augmented","slug":"lab-rag-label-boosted-retrieval-augmented","title":"LaB-RAG: Label Boosted Retrieval Augmented Generation for Radiology Report Generation","date":"2024-11-25","arxiv_id":"2411.16523","repositories_listed":1,"syntology":null},{"url":"/paper/fg-cxr-a-radiologist-aligned-gaze-dataset-for","slug":"fg-cxr-a-radiologist-aligned-gaze-dataset-for","title":"FG-CXR: A Radiologist-Aligned Gaze Dataset for Enhancing Interpretability in Chest X-Ray Report Generation","date":"2024-11-23","arxiv_id":"2411.15413","repositories_listed":1,"syntology":{"n":5,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/fg-cxr-a-radiologist-aligned-gaze-dataset-for#ran","syntology_url":"https://syntology.ai/paper/2411.15413","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.15413"}},"official":{"repos":["uark-aicv/fg-cxr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/lmm-driven-semantic-image-text-coding-for","slug":"lmm-driven-semantic-image-text-coding-for","title":"LMM-driven Semantic Image-Text Coding for Ultra Low-bitrate Learned Image Compression","date":"2024-11-20","arxiv_id":"2411.13033","repositories_listed":1,"syntology":null},{"url":"/paper/the-power-of-many-multi-agent-multimodal","slug":"the-power-of-many-multi-agent-multimodal","title":"The Power of Many: Multi-Agent Multimodal Models for Cultural Image Captioning","date":"2024-11-18","arxiv_id":"2411.11758","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-power-of-many-multi-agent-multimodal#ran","syntology_url":"https://syntology.ai/paper/2411.11758","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.11758"}},"official":{"repos":["michigannlp/mosaic"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bridging-the-visual-gap-fine-tuning","slug":"bridging-the-visual-gap-fine-tuning","title":"Bridging the Visual Gap: Fine-Tuning Multimodal Models with Knowledge-Adapted Captions","date":"2024-11-13","arxiv_id":"2411.09018","repositories_listed":1,"syntology":null},{"url":"/paper/llm2clip-powerful-language-model-unlock","slug":"llm2clip-powerful-language-model-unlock","title":"LLM2CLIP: Powerful Language Model Unlocks Richer Visual Representation","date":"2024-11-07","arxiv_id":"2411.04997","repositories_listed":1,"syntology":null},{"url":"/paper/precision-or-recall-an-analysis-of-image","slug":"precision-or-recall-an-analysis-of-image","title":"Precision or Recall? An Analysis of Image Captions for Training Text-to-Image Generation Model","date":"2024-11-07","arxiv_id":"2411.05079","repositories_listed":1,"syntology":null},{"url":"/paper/nearest-neighbor-normalization-improves","slug":"nearest-neighbor-normalization-improves","title":"Nearest Neighbor Normalization Improves Multimodal Retrieval","date":"2024-10-31","arxiv_id":"2410.24114","repositories_listed":1,"syntology":null},{"url":"/paper/adem-vl-adaptive-and-embedded-fusion-for","slug":"adem-vl-adaptive-and-embedded-fusion-for","title":"ADEM-VL: Adaptive and Embedded Fusion for Efficient Vision-Language Tuning","date":"2024-10-23","arxiv_id":"2410.17779","repositories_listed":1,"syntology":null},{"url":"/paper/altogether-image-captioning-via-re-aligning","slug":"altogether-image-captioning-via-re-aligning","title":"Altogether: Image Captioning via Re-aligning Alt-text","date":"2024-10-22","arxiv_id":"2410.17251","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/altogether-image-captioning-via-re-aligning#ran","syntology_url":"https://syntology.ai/paper/2410.17251","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.17251"}},"official":{"repos":["facebookresearch/metaclip"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/frontiers-in-intelligent-colonoscopy","slug":"frontiers-in-intelligent-colonoscopy","title":"Frontiers in Intelligent Colonoscopy","date":"2024-10-22","arxiv_id":"2410.17241","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/frontiers-in-intelligent-colonoscopy#ran","syntology_url":"https://syntology.ai/paper/2410.17241","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.17241"}},"official":{"repos":["ai4colonoscopy/intelliscope"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/an-efficient-system-for-automatic-map","slug":"an-efficient-system-for-automatic-map","title":"An Efficient System for Automatic Map Storytelling -- A Case Study on Historical Maps","date":"2024-10-21","arxiv_id":"2410.15780","repositories_listed":1,"syntology":null},{"url":"/paper/tips-text-image-pretraining-with-spatial","slug":"tips-text-image-pretraining-with-spatial","title":"TIPS: Text-Image Pretraining with Spatial Awareness","date":"2024-10-21","arxiv_id":"2410.16512","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/tips-text-image-pretraining-with-spatial#ran","syntology_url":"https://syntology.ai/paper/2410.16512","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.16512"}},"official":null}},{"url":"/paper/remember-retrieve-and-generate-understanding","slug":"remember-retrieve-and-generate-understanding","title":"RAP: Retrieval-Augmented Personalization for Multimodal Large Language Models","date":"2024-10-17","arxiv_id":"2410.13360","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":10,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/remember-retrieve-and-generate-understanding#ran","syntology_url":"https://syntology.ai/paper/2410.13360","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.13360"}},"official":{"repos":["hoar012/rap-mllm"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/self-adaptive-multimodal-retrieval-augmented","slug":"self-adaptive-multimodal-retrieval-augmented","title":"Self-adaptive Multimodal Retrieval-Augmented Generation","date":"2024-10-15","arxiv_id":"2410.11321","repositories_listed":1,"syntology":null},{"url":"/paper/a-unified-debiasing-approach-for-vision","slug":"a-unified-debiasing-approach-for-vision","title":"A Unified Debiasing Approach for Vision-Language Models across Modalities and Tasks","date":"2024-10-10","arxiv_id":"2410.07593","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/a-unified-debiasing-approach-for-vision#ran","syntology_url":"https://syntology.ai/paper/2410.07593","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.07593"}},"official":{"repos":["HoinJung/Unified-Debiaisng-VLM-SFID"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/an-eye-for-an-ear-zero-shot-audio-description","slug":"an-eye-for-an-ear-zero-shot-audio-description","title":"An Eye for an Ear: Zero-shot Audio Description Leveraging an Image Captioner using Audiovisual Distribution Alignment","date":"2024-10-08","arxiv_id":"2410.05997","repositories_listed":1,"syntology":null},{"url":"/paper/core-tokensets-for-data-efficient-sequential","slug":"core-tokensets-for-data-efficient-sequential","title":"Core Tokensets for Data-efficient Sequential Training of Transformers","date":"2024-10-08","arxiv_id":"2410.05800","repositories_listed":1,"syntology":null},{"url":"/paper/capeen-image-captioning-with-early-exits-and","slug":"capeen-image-captioning-with-early-exits-and","title":"CAPEEN: Image Captioning with Early Exits and Knowledge Distillation","date":"2024-10-06","arxiv_id":"2410.04433","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/capeen-image-captioning-with-early-exits-and#ran","syntology_url":"https://syntology.ai/paper/2410.04433","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.04433"}},"official":{"repos":["div290/capeen"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/trope-training-free-object-part-enhancement","slug":"trope-training-free-object-part-enhancement","title":"TROPE: TRaining-Free Object-Part Enhancement for Seamlessly Improving Fine-Grained Zero-Shot Image Captioning","date":"2024-09-30","arxiv_id":"2409.19960","repositories_listed":1,"syntology":null},{"url":"/paper/ifcap-image-like-retrieval-and-frequency","slug":"ifcap-image-like-retrieval-and-frequency","title":"IFCap: Image-like Retrieval and Frequency-based Entity Filtering for Zero-shot Captioning","date":"2024-09-26","arxiv_id":"2409.18046","repositories_listed":1,"syntology":null},{"url":"/paper/molmo-and-pixmo-open-weights-and-open-data","slug":"molmo-and-pixmo-open-weights-and-open-data","title":"Molmo and PixMo: Open Weights and Open Data for State-of-the-Art Vision-Language Models","date":"2024-09-25","arxiv_id":"2409.17146","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/molmo-and-pixmo-open-weights-and-open-data#ran","syntology_url":"https://syntology.ai/paper/2409.17146","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.17146"}},"official":{"repos":["allenai/molmo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/2409-13407","slug":"2409-13407","title":"Instruction-guided Multi-Granularity Segmentation and Captioning with Large Multimodal Model","date":"2024-09-20","arxiv_id":"2409.13407","repositories_listed":1,"syntology":null},{"url":"/paper/yesbut-a-high-quality-annotated-multimodal","slug":"yesbut-a-high-quality-annotated-multimodal","title":"YesBut: A High-Quality Annotated Multimodal Dataset for evaluating Satire Comprehension capability of Vision-Language Models","date":"2024-09-20","arxiv_id":"2409.13592","repositories_listed":1,"syntology":null},{"url":"/paper/journeybench-a-challenging-one-stop-vision","slug":"journeybench-a-challenging-one-stop-vision","title":"JourneyBench: A Challenging One-Stop Vision-Language Understanding Benchmark of Generated Images","date":"2024-09-19","arxiv_id":"2409.12953","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/journeybench-a-challenging-one-stop-vision#ran","syntology_url":"https://syntology.ai/paper/2409.12953","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.12953"}},"official":{"repos":["journeybench/journeybench"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/kale-an-artwork-image-captioning-system","slug":"kale-an-artwork-image-captioning-system","title":"KALE: An Artwork Image Captioning System Augmented with Heterogeneous Graph","date":"2024-09-17","arxiv_id":"2409.10921","repositories_listed":1,"syntology":null},{"url":"/paper/playground-v3-improving-text-to-image","slug":"playground-v3-improving-text-to-image","title":"Playground v3: Improving Text-to-Image Alignment with Deep-Fusion Large Language Models","date":"2024-09-16","arxiv_id":"2409.10695","repositories_listed":1,"syntology":{"n":21,"n_ran":18,"n_constructed":0,"n_ran_checked":9,"n_instrument":9,"n_unverified":3,"n_honours":4,"n_violates":1,"n_no_contract":4,"n_pointer_only":21,"phrase":"18 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 4 honoured, 1 violated, 4 with no contract checked; 9 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/playground-v3-improving-text-to-image#ran","syntology_url":"https://syntology.ai/paper/2409.10695","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.10695"}},"official":null}},{"url":"/paper/training-free-zs-cir-via-weighted-modality","slug":"training-free-zs-cir-via-weighted-modality","title":"Training-free Zero-shot Composed Image Retrieval via Weighted Modality Fusion and Similarity","date":"2024-09-07","arxiv_id":"2409.04918","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/training-free-zs-cir-via-weighted-modality#ran","syntology_url":"https://syntology.ai/paper/2409.04918","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.04918"}},"official":{"repos":["whats2000/WeiMoCIR"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/kvasir-vqa-a-text-image-pair-gi-tract-dataset","slug":"kvasir-vqa-a-text-image-pair-gi-tract-dataset","title":"Kvasir-VQA: A Text-Image Pair GI Tract Dataset","date":"2024-09-02","arxiv_id":"2409.01437","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"0 ran · 3 unverified","sample_list":"/paper/kvasir-vqa-a-text-image-pair-gi-tract-dataset#ran","syntology_url":"https://syntology.ai/paper/2409.01437","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.01437"}},"official":{"repos":["simula/Kvasir-VQA"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/multimath-bridging-visual-and-mathematical","slug":"multimath-bridging-visual-and-mathematical","title":"MultiMath: Bridging Visual and Mathematical Reasoning for Large Language Models","date":"2024-08-30","arxiv_id":"2409.00147","repositories_listed":1,"syntology":{"n":18,"n_ran":17,"n_constructed":0,"n_ran_checked":9,"n_instrument":8,"n_unverified":1,"n_honours":0,"n_violates":3,"n_no_contract":6,"n_pointer_only":3,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 3 violated, 6 with no contract checked; 8 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multimath-bridging-visual-and-mathematical#ran","syntology_url":"https://syntology.ai/paper/2409.00147","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.00147"}},"official":{"repos":["pengshuai-rin/multimath"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/see-or-guess-counterfactually-regularized","slug":"see-or-guess-counterfactually-regularized","title":"See or Guess: Counterfactually Regularized Image Captioning","date":"2024-08-29","arxiv_id":"2408.16809","repositories_listed":1,"syntology":null},{"url":"/paper/revisiting-image-captioning-training-paradigm","slug":"revisiting-image-captioning-training-paradigm","title":"Revisiting Image Captioning Training Paradigm via Direct CLIP-based Optimization","date":"2024-08-26","arxiv_id":"2408.14547","repositories_listed":1,"syntology":null},{"url":"/paper/fuse-ing-language-models-zero-shot-adapter","slug":"fuse-ing-language-models-zero-shot-adapter","title":"FUSE-ing Language Models: Zero-Shot Adapter Discovery for Prompt Optimization Across Tokenizers","date":"2024-08-09","arxiv_id":"2408.04816","repositories_listed":1,"syntology":null},{"url":"/paper/bridge-bridging-gaps-in-image-captioning","slug":"bridge-bridging-gaps-in-image-captioning","title":"BRIDGE: Bridging Gaps in Image Captioning Evaluation with Stronger Visual Cues","date":"2024-07-29","arxiv_id":"2407.20341","repositories_listed":1,"syntology":null},{"url":"/paper/swift-semantic-watermarking-for-image-forgery","slug":"swift-semantic-watermarking-for-image-forgery","title":"SWIFT: Semantic Watermarking for Image Forgery Thwarting","date":"2024-07-26","arxiv_id":"2407.18995","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":1,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/swift-semantic-watermarking-for-image-forgery#ran","syntology_url":"https://syntology.ai/paper/2407.18995","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.18995"}},"official":{"repos":["gautierevn/swift_watermarking"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/diffx-guide-your-layout-to-cross-modal","slug":"diffx-guide-your-layout-to-cross-modal","title":"DiffX: Guide Your Layout to Cross-Modal Generative Modeling","date":"2024-07-22","arxiv_id":"2407.15488","repositories_listed":1,"syntology":null},{"url":"/paper/cic-bart-ssa-controllable-image-captioning","slug":"cic-bart-ssa-controllable-image-captioning","title":"CIC-BART-SSA: Controllable Image Captioning with Structured Semantic Augmentation","date":"2024-07-16","arxiv_id":"2407.11393","repositories_listed":1,"syntology":null},{"url":"/paper/controllable-contextualized-image-captioning","slug":"controllable-contextualized-image-captioning","title":"Controllable Contextualized Image Captioning: Directing the Visual Narrative through User-Defined Highlights","date":"2024-07-16","arxiv_id":"2407.11449","repositories_listed":1,"syntology":null},{"url":"/paper/avcap-leveraging-audio-visual-features-as","slug":"avcap-leveraging-audio-visual-features-as","title":"AVCap: Leveraging Audio-Visual Features as Text Tokens for Captioning","date":"2024-07-10","arxiv_id":"2407.07801","repositories_listed":1,"syntology":null},{"url":"/paper/pseudo-ris-distinctive-pseudo-supervision","slug":"pseudo-ris-distinctive-pseudo-supervision","title":"Pseudo-RIS: Distinctive Pseudo-supervision Generation for Referring Image Segmentation","date":"2024-07-10","arxiv_id":"2407.07412","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-image-captions-for-selective-whole","slug":"leveraging-image-captions-for-selective-whole","title":"Leveraging image captions for selective whole slide image annotation","date":"2024-07-08","arxiv_id":"2407.06363","repositories_listed":1,"syntology":null},{"url":"/paper/mfc-bench-benchmarking-multimodal-fact","slug":"mfc-bench-benchmarking-multimodal-fact","title":"MFC-Bench: Benchmarking Multimodal Fact-Checking with Large Vision-Language Models","date":"2024-06-17","arxiv_id":"2406.11288","repositories_listed":1,"syntology":null},{"url":"/paper/imagenet3d-towards-general-purpose-object","slug":"imagenet3d-towards-general-purpose-object","title":"ImageNet3D: Towards General-Purpose Object-Level 3D Understanding","date":"2024-06-13","arxiv_id":"2406.09613","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/imagenet3d-towards-general-purpose-object#ran","syntology_url":"https://syntology.ai/paper/2406.09613","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09613"}},"official":{"repos":["wufeim/imagenet3d"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-vision-language-geo-foundation-model","slug":"towards-vision-language-geo-foundation-model","title":"Towards Vision-Language Geo-Foundation Model: A Survey","date":"2024-06-13","arxiv_id":"2406.09385","repositories_listed":1,"syntology":null},{"url":"/paper/yo-llava-your-personalized-language-and","slug":"yo-llava-your-personalized-language-and","title":"Yo'LLaVA: Your Personalized Language and Vision Assistant","date":"2024-06-13","arxiv_id":"2406.09400","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/yo-llava-your-personalized-language-and#ran","syntology_url":"https://syntology.ai/paper/2406.09400","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09400"}},"official":{"repos":["WisconsinAIVision/YoLLaVA"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/translating-speech-with-just-images","slug":"translating-speech-with-just-images","title":"Translating speech with just images","date":"2024-06-11","arxiv_id":"2406.07133","repositories_listed":1,"syntology":null},{"url":"/paper/fleur-an-explainable-reference-free","slug":"fleur-an-explainable-reference-free","title":"FLEUR: An Explainable Reference-Free Evaluation Metric for Image Captioning Using a Large Multimodal Model","date":"2024-06-10","arxiv_id":"2406.06004","repositories_listed":1,"syntology":null},{"url":"/paper/from-redundancy-to-relevance-enhancing","slug":"from-redundancy-to-relevance-enhancing","title":"From Redundancy to Relevance: Information Flow in LVLMs Across Reasoning Tasks","date":"2024-06-04","arxiv_id":"2406.06579","repositories_listed":1,"syntology":null},{"url":"/paper/understanding-retrieval-robustness-for","slug":"understanding-retrieval-robustness-for","title":"Understanding Retrieval Robustness for Retrieval-Augmented Image Captioning","date":"2024-06-04","arxiv_id":"2406.02265","repositories_listed":1,"syntology":null},{"url":"/paper/dragonfly-multi-resolution-zoom-supercharges","slug":"dragonfly-multi-resolution-zoom-supercharges","title":"Dragonfly: Multi-Resolution Zoom-In Encoding Enhances Vision-Language Models","date":"2024-06-03","arxiv_id":"2406.00977","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dragonfly-multi-resolution-zoom-supercharges#ran","syntology_url":"https://syntology.ai/paper/2406.00977","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.00977"}},"official":{"repos":["togethercomputer/dragonfly"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/image-captioning-via-dynamic-path","slug":"image-captioning-via-dynamic-path","title":"Image Captioning via Dynamic Path Customization","date":"2024-06-01","arxiv_id":"2406.00334","repositories_listed":1,"syntology":null},{"url":"/paper/rtgen-generating-region-text-pairs-for-open","slug":"rtgen-generating-region-text-pairs-for-open","title":"RTGen: Generating Region-Text Pairs for Open-Vocabulary Object Detection","date":"2024-05-30","arxiv_id":"2405.19854","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-and-improving-detail-image","slug":"benchmarking-and-improving-detail-image","title":"Benchmarking and Improving Detail Image Caption","date":"2024-05-29","arxiv_id":"2405.19092","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-and-improving-detail-image#ran","syntology_url":"https://syntology.ai/paper/2405.19092","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19092"}},"official":{"repos":["foundation-multimodal-models/capture"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-survey-on-vision-language-action-models-for","slug":"a-survey-on-vision-language-action-models-for","title":"A Survey on Vision-Language-Action Models for Embodied AI","date":"2024-05-23","arxiv_id":"2405.14093","repositories_listed":1,"syntology":null},{"url":"/paper/class-conditional-self-reward-mechanism-for","slug":"class-conditional-self-reward-mechanism-for","title":"Class-Conditional self-reward mechanism for improved Text-to-Image models","date":"2024-05-22","arxiv_id":"2405.13473","repositories_listed":1,"syntology":null},{"url":"/paper/unirag-universal-retrieval-augmentation-for","slug":"unirag-universal-retrieval-augmentation-for","title":"UniRAG: Universal Retrieval Augmentation for Large Vision Language Models","date":"2024-05-16","arxiv_id":"2405.10311","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":13,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/unirag-universal-retrieval-augmentation-for#ran","syntology_url":"https://syntology.ai/paper/2405.10311","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.10311"}},"official":{"repos":["castorini/unirag"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/boostlet-js-image-processing-plugins-for-the","slug":"boostlet-js-image-processing-plugins-for-the","title":"Boostlet.js: Image processing plugins for the web via JavaScript injection","date":"2024-05-13","arxiv_id":"2405.07868","repositories_listed":1,"syntology":null},{"url":"/paper/cumo-scaling-multimodal-llm-with-co-upcycled","slug":"cumo-scaling-multimodal-llm-with-co-upcycled","title":"CuMo: Scaling Multimodal LLM with Co-Upcycled Mixture-of-Experts","date":"2024-05-09","arxiv_id":"2405.05949","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":6,"n_instrument":5,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cumo-scaling-multimodal-llm-with-co-upcycled#ran","syntology_url":"https://syntology.ai/paper/2405.05949","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.05949"}},"official":{"repos":["shi-labs/cumo"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/llm-as-dataset-analyst-subpopulation","slug":"llm-as-dataset-analyst-subpopulation","title":"LLM as Dataset Analyst: Subpopulation Structure Discovery with Large Language Model","date":"2024-05-03","arxiv_id":"2405.02363","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/llm-as-dataset-analyst-subpopulation#ran","syntology_url":"https://syntology.ai/paper/2405.02363","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.02363"}},"official":{"repos":["llm-as-dataset-analyst/SSDLLM"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/technical-report-of-nice-challenge-at-cvpr","slug":"technical-report-of-nice-challenge-at-cvpr","title":"Technical Report of NICE Challenge at CVPR 2024: Caption Re-ranking Evaluation Using Ensembled CLIP and Consensus Scores","date":"2024-05-02","arxiv_id":"2405.01028","repositories_listed":1,"syntology":null},{"url":"/paper/omnisearchsage-multi-task-multi-entity","slug":"omnisearchsage-multi-task-multi-entity","title":"OmniSearchSage: Multi-Task Multi-Entity Embeddings for Pinterest Search","date":"2024-04-25","arxiv_id":"2404.16260","repositories_listed":1,"syntology":null},{"url":"/paper/lost-in-space-probing-fine-grained-spatial","slug":"lost-in-space-probing-fine-grained-spatial","title":"Lost in Space: Probing Fine-grained Spatial Understanding in Vision and Language Resamplers","date":"2024-04-21","arxiv_id":"2404.13594","repositories_listed":1,"syntology":null},{"url":"/paper/ladic-are-diffusion-models-really-inferior-to","slug":"ladic-are-diffusion-models-really-inferior-to","title":"LaDiC: Are Diffusion Models Really Inferior to Autoregressive Counterparts for Image-to-Text Generation?","date":"2024-04-16","arxiv_id":"2404.10763","repositories_listed":1,"syntology":null},{"url":"/paper/anchor-llm-driven-news-subject-conditioning","slug":"anchor-llm-driven-news-subject-conditioning","title":"ANCHOR: LLM-driven News Subject Conditioning for Text-to-Image Synthesis","date":"2024-04-15","arxiv_id":"2404.10141","repositories_listed":1,"syntology":null},{"url":"/paper/bridging-vision-and-language-spaces-with","slug":"bridging-vision-and-language-spaces-with","title":"Bridging Vision and Language Spaces with Assignment Prediction","date":"2024-04-15","arxiv_id":"2404.09632","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bridging-vision-and-language-spaces-with#ran","syntology_url":"https://syntology.ai/paper/2404.09632","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.09632"}},"official":{"repos":["park-jungin/vlap"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-visual-question-answering-through","slug":"enhancing-visual-question-answering-through","title":"Enhancing Visual Question Answering through Question-Driven Image Captions as Prompts","date":"2024-04-12","arxiv_id":"2404.08589","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/enhancing-visual-question-answering-through#ran","syntology_url":"https://syntology.ai/paper/2404.08589","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.08589"}},"official":{"repos":["ovguyo/captions-in-vqa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/flora-enhancing-vision-language-models-with","slug":"flora-enhancing-vision-language-models-with","title":"FLoRA: Enhancing Vision-Language Models with Parameter-Efficient Federated Learning","date":"2024-04-12","arxiv_id":"2404.15182","repositories_listed":1,"syntology":null},{"url":"/paper/harnessing-the-power-of-large-vision-language","slug":"harnessing-the-power-of-large-vision-language","title":"Harnessing the Power of Large Vision Language Models for Synthetic Image Detection","date":"2024-04-03","arxiv_id":"2404.02726","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/harnessing-the-power-of-large-vision-language#ran","syntology_url":"https://syntology.ai/paper/2404.02726","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.02726"}},"official":{"repos":["mamadou-keita/vlm-detect"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/bi-lora-a-vision-language-approach-for","slug":"bi-lora-a-vision-language-approach-for","title":"Bi-LORA: A Vision-Language Approach for Synthetic Image Detection","date":"2024-04-02","arxiv_id":"2404.01959","repositories_listed":1,"syntology":null},{"url":"/paper/disentangled-pre-training-for-human-object","slug":"disentangled-pre-training-for-human-object","title":"Disentangled Pre-training for Human-Object Interaction Detection","date":"2024-04-02","arxiv_id":"2404.01725","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/disentangled-pre-training-for-human-object#ran","syntology_url":"https://syntology.ai/paper/2404.01725","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.01725"}},"official":{"repos":["xingaoli/dp-hoi"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-by-correction-efficient-tuning-task","slug":"learning-by-correction-efficient-tuning-task","title":"Learning by Correction: Efficient Tuning Task for Zero-Shot Generative Vision-Language Reasoning","date":"2024-04-01","arxiv_id":"2404.00909","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":1,"n_ran_checked":4,"n_instrument":4,"n_unverified":2,"n_honours":1,"n_violates":2,"n_no_contract":1,"n_pointer_only":0,"phrase":"8 ran (of which 1 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 2 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/learning-by-correction-efficient-tuning-task#ran","syntology_url":"https://syntology.ai/paper/2404.00909","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.00909"}},"official":{"repos":["shtuplus/iccc_cvpr2024"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":1,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/locca-visual-pretraining-with-location-aware","slug":"locca-visual-pretraining-with-location-aware","title":"LocCa: Visual Pretraining with Location-aware Captioners","date":"2024-03-28","arxiv_id":"2403.19596","repositories_listed":1,"syntology":null},{"url":"/paper/semantic-map-based-generation-of-navigation","slug":"semantic-map-based-generation-of-navigation","title":"Semantic Map-based Generation of Navigation Instructions","date":"2024-03-28","arxiv_id":"2403.19603","repositories_listed":1,"syntology":null},{"url":"/paper/can-language-beat-numerical-regression","slug":"can-language-beat-numerical-regression","title":"Can Language Beat Numerical Regression? Language-Based Multimodal Trajectory Prediction","date":"2024-03-27","arxiv_id":"2403.18447","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":13,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/can-language-beat-numerical-regression#ran","syntology_url":"https://syntology.ai/paper/2403.18447","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.18447"}},"official":{"repos":["inhwanbae/lmtrajectory"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/cognitive-resilience-unraveling-the","slug":"cognitive-resilience-unraveling-the","title":"Cognitive resilience: Unraveling the proficiency of image-captioning models to interpret masked visual content","date":"2024-03-23","arxiv_id":"2403.15876","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cognitive-resilience-unraveling-the#ran","syntology_url":"https://syntology.ai/paper/2403.15876","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.15876"}},"official":{"repos":["dodoxxb/cognitive-resilience"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/boosting-transferability-in-vision-language","slug":"boosting-transferability-in-vision-language","title":"Boosting Transferability in Vision-Language Attacks via Diversification along the Intersection Region of Adversarial Trajectory","date":"2024-03-19","arxiv_id":"2403.12445","repositories_listed":1,"syntology":null},{"url":"/paper/vl-icl-bench-the-devil-in-the-details-of","slug":"vl-icl-bench-the-devil-in-the-details-of","title":"VL-ICL Bench: The Devil in the Details of Multimodal In-Context Learning","date":"2024-03-19","arxiv_id":"2403.13164","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/vl-icl-bench-the-devil-in-the-details-of#ran","syntology_url":"https://syntology.ai/paper/2403.13164","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.13164"}},"official":{"repos":["ys-zong/vl-icl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/does-the-performance-of-text-to-image","slug":"does-the-performance-of-text-to-image","title":"Does the Performance of Text-to-Image Retrieval Models Generalize Beyond Captions-as-a-Query?","date":"2024-03-15","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/are-vision-language-models-texture-or-shape","slug":"are-vision-language-models-texture-or-shape","title":"Can We Talk Models Into Seeing the World Differently?","date":"2024-03-14","arxiv_id":"2403.09193","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/are-vision-language-models-texture-or-shape#ran","syntology_url":"https://syntology.ai/paper/2403.09193","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.09193"}},"official":{"repos":["paulgavrikov/vlm_shapebias"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/beyond-text-frozen-large-language-models-in","slug":"beyond-text-frozen-large-language-models-in","title":"Beyond Text: Frozen Large Language Models in Visual Signal Comprehension","date":"2024-03-12","arxiv_id":"2403.07874","repositories_listed":1,"syntology":{"n":26,"n_ran":17,"n_constructed":0,"n_ran_checked":6,"n_instrument":11,"n_unverified":9,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":26,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 11 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/beyond-text-frozen-large-language-models-in#ran","syntology_url":"https://syntology.ai/paper/2403.07874","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.07874"}},"official":{"repos":["zh460045050/v2l-tokenizer"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/pix2pix-onthefly-leveraging-llms-for","slug":"pix2pix-onthefly-leveraging-llms-for","title":"Leveraging LLMs for On-the-Fly Instruction Guided Image Editing","date":"2024-03-12","arxiv_id":"2403.08004","repositories_listed":1,"syntology":null}],"record_sha256":"eca00e4eca3d1b07b283ba4758d2783be806c01b3393a255c77a382d2def7059","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}