{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/image-captioning/papers/4","list_of":"/task/image-captioning","task":"Image Captioning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":4,"pages_in_order":19,"rows_per_page":100,"rows":[301,400],"of":1878,"counts":{"archive_papers_tagged":1878,"with_a_code_link":774,"where_syntology_ran_a_sample":243,"not_listed_spam_title":0,"listed":1878,"listed_where_code_ran":243,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":201,"every_run_a_failure_of_syntologys_instrument":42,"listed_with_a_run_with_no_instrument_failure":201,"listed_every_run_a_failure_of_syntologys_instrument":42,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/image-captioning","prev":"/task/image-captioning/papers/3","next":"/task/image-captioning/papers/5","papers":[{"url":"/paper/meacap-memory-augmented-zero-shot-image","slug":"meacap-memory-augmented-zero-shot-image","title":"MeaCap: Memory-Augmented Zero-shot Image Captioning","date":"2024-03-06","arxiv_id":"2403.03715","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":1,"n_ran_checked":4,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":10,"phrase":"6 ran (of which 1 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/meacap-memory-augmented-zero-shot-image#ran","syntology_url":"https://syntology.ai/paper/2403.03715","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.03715"}},"official":{"repos":["joeyz0z/meacap"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":1,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/differentially-private-representation","slug":"differentially-private-representation","title":"Differentially Private Representation Learning via Image Captioning","date":"2024-03-04","arxiv_id":"2403.02506","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/differentially-private-representation#ran","syntology_url":"https://syntology.ai/paper/2403.02506","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.02506"}},"official":{"repos":["facebookresearch/dpcap"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vtg-gpt-tuning-free-zero-shot-video-temporal-1","slug":"vtg-gpt-tuning-free-zero-shot-video-temporal-1","title":"VTG-GPT: Tuning-Free Zero-Shot Video Temporal Grounding with GPT","date":"2024-03-04","arxiv_id":"2403.02076","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":4,"n_instrument":5,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vtg-gpt-tuning-free-zero-shot-video-temporal-1#ran","syntology_url":"https://syntology.ai/paper/2403.02076","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.02076"}},"official":{"repos":["YoucanBaby/VTG-GPT"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/what-is-missing-in-multilingual-visual","slug":"what-is-missing-in-multilingual-visual","title":"What Is Missing in Multilingual Visual Reasoning and How to Fix It","date":"2024-03-03","arxiv_id":"2403.01404","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/what-is-missing-in-multilingual-visual#ran","syntology_url":"https://syntology.ai/paper/2403.01404","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.01404"}},"official":{"repos":["yueqis/multilingual_visual_reasoning"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-explicit-spatial-relationships-in","slug":"improving-explicit-spatial-relationships-in","title":"Improving Explicit Spatial Relationships in Text-to-Image Generation through an Automatically Derived Dataset","date":"2024-03-01","arxiv_id":"2403.00587","repositories_listed":1,"syntology":null},{"url":"/paper/polos-multimodal-metric-learning-from-human","slug":"polos-multimodal-metric-learning-from-human","title":"Polos: Multimodal Metric Learning from Human Feedback for Image Captioning","date":"2024-02-28","arxiv_id":"2402.18091","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/polos-multimodal-metric-learning-from-human#ran","syntology_url":"https://syntology.ai/paper/2402.18091","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.18091"}},"official":{"repos":["keio-smilab24/Polos"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/distinctive-image-captioning-leveraging","slug":"distinctive-image-captioning-leveraging","title":"Distinctive Image Captioning: Leveraging Ground Truth Captions in CLIP Guided Reinforcement Learning","date":"2024-02-21","arxiv_id":"2402.13936","repositories_listed":1,"syntology":null},{"url":"/paper/aicattack-adversarial-image-captioning-attack","slug":"aicattack-adversarial-image-captioning-attack","title":"AICAttack: Adversarial Image Captioning Attack with Attention-Based Optimization","date":"2024-02-19","arxiv_id":"2402.11940","repositories_listed":1,"syntology":null},{"url":"/paper/chatearthnet-a-global-scale-high-quality","slug":"chatearthnet-a-global-scale-high-quality","title":"ChatEarthNet: A Global-Scale Image-Text Dataset Empowering Vision-Language Geo-Foundation Models","date":"2024-02-17","arxiv_id":"2402.11325","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/chatearthnet-a-global-scale-high-quality#ran","syntology_url":"https://syntology.ai/paper/2402.11325","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11325"}},"official":{"repos":["zhu-xlab/ChatEarthNet"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/examining-gender-and-racial-bias-in-large","slug":"examining-gender-and-racial-bias-in-large","title":"Examining Gender and Racial Bias in Large Vision-Language Models Using a Novel Dataset of Parallel Images","date":"2024-02-08","arxiv_id":"2402.05779","repositories_listed":1,"syntology":null},{"url":"/paper/gpts-are-multilingual-annotators-for-sequence","slug":"gpts-are-multilingual-annotators-for-sequence","title":"GPTs Are Multilingual Annotators for Sequence Generation Tasks","date":"2024-02-08","arxiv_id":"2402.05512","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/gpts-are-multilingual-annotators-for-sequence#ran","syntology_url":"https://syntology.ai/paper/2402.05512","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05512"}},"official":{"repos":["c-juhwan/gpt-multilingual-annotator"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/text-guided-image-clustering","slug":"text-guided-image-clustering","title":"Text-Guided Image Clustering","date":"2024-02-05","arxiv_id":"2402.02996","repositories_listed":1,"syntology":null},{"url":"/paper/good-at-captioning-bad-at-counting","slug":"good-at-captioning-bad-at-counting","title":"Good at captioning, bad at counting: Benchmarking GPT-4V on Earth observation data","date":"2024-01-31","arxiv_id":"2401.17600","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/good-at-captioning-bad-at-counting#ran","syntology_url":"https://syntology.ai/paper/2401.17600","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.17600"}},"official":{"repos":["Earth-Intelligence-Lab/vleo-bench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/scimmir-benchmarking-scientific-multi-modal","slug":"scimmir-benchmarking-scientific-multi-modal","title":"SciMMIR: Benchmarking Scientific Multi-modal Information Retrieval","date":"2024-01-24","arxiv_id":"2401.13478","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":10,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/scimmir-benchmarking-scientific-multi-modal#ran","syntology_url":"https://syntology.ai/paper/2401.13478","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.13478"}},"official":{"repos":["wusiwei0410/scimmir"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/veagle-advancements-in-multimodal","slug":"veagle-advancements-in-multimodal","title":"Veagle: Advancements in Multimodal Representation Learning","date":"2024-01-18","arxiv_id":"2403.08773","repositories_listed":1,"syntology":null},{"url":"/paper/hyperparameter-free-approach-for-faster","slug":"hyperparameter-free-approach-for-faster","title":"Hyperparameter-Free Approach for Faster Minimum Bayes Risk Decoding","date":"2024-01-05","arxiv_id":"2401.02749","repositories_listed":1,"syntology":null},{"url":"/paper/mining-fine-grained-image-text-alignment-for","slug":"mining-fine-grained-image-text-alignment-for","title":"Mining Fine-Grained Image-Text Alignment for Zero-Shot Captioning via Text-Only Training","date":"2024-01-04","arxiv_id":"2401.02347","repositories_listed":1,"syntology":{"n":8,"n_ran":3,"n_constructed":1,"n_ran_checked":1,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":8,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/mining-fine-grained-image-text-alignment-for#ran","syntology_url":"https://syntology.ai/paper/2401.02347","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.02347"}},"official":{"repos":["artanic30/maccap"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":5,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/gpt-4v-ision-is-a-generalist-web-agent-if","slug":"gpt-4v-ision-is-a-generalist-web-agent-if","title":"GPT-4V(ision) is a Generalist Web Agent, if Grounded","date":"2024-01-03","arxiv_id":"2401.01614","repositories_listed":1,"syntology":{"n":14,"n_ran":13,"n_constructed":0,"n_ran_checked":12,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":14,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/gpt-4v-ision-is-a-generalist-web-agent-if#ran","syntology_url":"https://syntology.ai/paper/2401.01614","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.01614"}},"official":{"repos":["osu-nlp-group/seeact"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/vcoder-versatile-vision-encoders-for","slug":"vcoder-versatile-vision-encoders-for","title":"VCoder: Versatile Vision Encoders for Multimodal Large Language Models","date":"2023-12-21","arxiv_id":"2312.14233","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vcoder-versatile-vision-encoders-for#ran","syntology_url":"https://syntology.ai/paper/2312.14233","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.14233"}},"official":{"repos":["shi-labs/vcoder"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/p-laplacian-adaptation-for-generative-pre","slug":"p-laplacian-adaptation-for-generative-pre","title":"p-Laplacian Adaptation for Generative Pre-trained Vision-Language Models","date":"2023-12-17","arxiv_id":"2312.10613","repositories_listed":1,"syntology":null},{"url":"/paper/a-picture-is-worth-more-than-77-text-tokens","slug":"a-picture-is-worth-more-than-77-text-tokens","title":"A Picture is Worth More Than 77 Text Tokens: Evaluating CLIP-Style Models on Dense Captions","date":"2023-12-14","arxiv_id":"2312.08578","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":12,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/a-picture-is-worth-more-than-77-text-tokens#ran","syntology_url":"https://syntology.ai/paper/2312.08578","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.08578"}},"official":{"repos":["facebookresearch/dci"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/vl-gpt-a-generative-pre-trained-transformer","slug":"vl-gpt-a-generative-pre-trained-transformer","title":"VL-GPT: A Generative Pre-trained Transformer for Vision and Language Understanding and Generation","date":"2023-12-14","arxiv_id":"2312.09251","repositories_listed":1,"syntology":null},{"url":"/paper/genixer-empowering-multimodal-large-language","slug":"genixer-empowering-multimodal-large-language","title":"Genixer: Empowering Multimodal Large Language Models as a Powerful Data Generator","date":"2023-12-11","arxiv_id":"2312.06731","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":4,"n_pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 1 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/genixer-empowering-multimodal-large-language#ran","syntology_url":"https://syntology.ai/paper/2312.06731","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.06731"}},"official":{"repos":["zhaohengyuan1/genixer"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unifying-text-tables-and-images-for","slug":"unifying-text-tables-and-images-for","title":"Unifying Text, Tables, and Images for Multimodal Question Answering","date":"2023-12-10","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/pixlore-a-dataset-driven-approach-to-rich","slug":"pixlore-a-dataset-driven-approach-to-rich","title":"PixLore: A Dataset-driven Approach to Rich Image Captioning","date":"2023-12-08","arxiv_id":"2312.05349","repositories_listed":1,"syntology":null},{"url":"/paper/mocha-multi-objective-reinforcement","slug":"mocha-multi-objective-reinforcement","title":"Mitigating Open-Vocabulary Caption Hallucinations","date":"2023-12-06","arxiv_id":"2312.03631","repositories_listed":1,"syntology":{"n":15,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":15,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/mocha-multi-objective-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2312.03631","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.03631"}},"official":{"repos":["assafbk/mocha_code"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/automatic-report-generation-for-1","slug":"automatic-report-generation-for-1","title":"Automatic Report Generation for Histopathology images using pre-trained Vision Transformers and BERT","date":"2023-12-03","arxiv_id":"2312.01435","repositories_listed":1,"syntology":null},{"url":"/paper/bootstrapping-interactive-image-text","slug":"bootstrapping-interactive-image-text","title":"Bootstrapping Interactive Image-Text Alignment for Remote Sensing Image Captioning","date":"2023-12-02","arxiv_id":"2312.01191","repositories_listed":1,"syntology":null},{"url":"/paper/video-summarization-towards-entity-aware","slug":"video-summarization-towards-entity-aware","title":"Video Summarization: Towards Entity-Aware Captions","date":"2023-12-01","arxiv_id":"2312.02188","repositories_listed":1,"syntology":null},{"url":"/paper/contrastive-vision-language-alignment-makes","slug":"contrastive-vision-language-alignment-makes","title":"Contrastive Vision-Language Alignment Makes Efficient Instruction Learner","date":"2023-11-29","arxiv_id":"2311.17945","repositories_listed":1,"syntology":null},{"url":"/paper/plug-and-play-dense-label-free-extraction-of","slug":"plug-and-play-dense-label-free-extraction-of","title":"Emergent Open-Vocabulary Semantic Segmentation from Off-the-shelf Vision-Language Models","date":"2023-11-28","arxiv_id":"2311.17095","repositories_listed":1,"syntology":null},{"url":"/paper/zero-shot-audio-captioning-with-audio","slug":"zero-shot-audio-captioning-with-audio","title":"Zero-shot audio captioning with audio-language model guidance and audio context keywords","date":"2023-11-14","arxiv_id":"2311.08396","repositories_listed":1,"syntology":{"n":17,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":17,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/zero-shot-audio-captioning-with-audio#ran","syntology_url":"https://syntology.ai/paper/2311.08396","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.08396"}},"official":{"repos":["explainableml/zeraucap"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/monkey-image-resolution-and-text-label-are","slug":"monkey-image-resolution-and-text-label-are","title":"Monkey: Image Resolution and Text Label Are Important Things for Large Multi-modal Models","date":"2023-11-11","arxiv_id":"2311.06607","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/monkey-image-resolution-and-text-label-are#ran","syntology_url":"https://syntology.ai/paper/2311.06607","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.06607"}},"official":{"repos":["yuliang-liu/monkey"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/zero-shot-translation-of-attention-patterns","slug":"zero-shot-translation-of-attention-patterns","title":"Zero-shot Translation of Attention Patterns in VQA Models to Natural Language","date":"2023-11-08","arxiv_id":"2311.05043","repositories_listed":1,"syntology":null},{"url":"/paper/deeppatent2-a-large-scale-benchmarking-corpus","slug":"deeppatent2-a-large-scale-benchmarking-corpus","title":"DeepPatent2: A Large-Scale Benchmarking Corpus for Technical Drawing Understanding","date":"2023-11-07","arxiv_id":"2311.04098","repositories_listed":1,"syntology":null},{"url":"/paper/jaspice-automatic-evaluation-metric-using","slug":"jaspice-automatic-evaluation-metric-using","title":"JaSPICE: Automatic Evaluation Metric Using Predicate-Argument Structures for Image Captioning Models","date":"2023-11-07","arxiv_id":"2311.04192","repositories_listed":1,"syntology":null},{"url":"/paper/glamm-pixel-grounding-large-multimodal-model","slug":"glamm-pixel-grounding-large-multimodal-model","title":"GLaMM: Pixel Grounding Large Multimodal Model","date":"2023-11-06","arxiv_id":"2311.03356","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/glamm-pixel-grounding-large-multimodal-model#ran","syntology_url":"https://syntology.ai/paper/2311.03356","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.03356"}},"official":{"repos":["mbzuai-oryx/groundingLMM"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/neusyre-neuro-symbolic-visual-understanding","slug":"neusyre-neuro-symbolic-visual-understanding","title":"NeuSyRE: Neuro-Symbolic Visual Understanding and Reasoning Framework based on Scene Graph Enrichment","date":"2023-11-05","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/sam-guided-enhanced-fine-grained-encoding","slug":"sam-guided-enhanced-fine-grained-encoding","title":"Sam-Guided Enhanced Fine-Grained Encoding with Mixed Semantic Learning for Medical Image Captioning","date":"2023-11-02","arxiv_id":"2311.01004","repositories_listed":1,"syntology":null},{"url":"/paper/language-guided-visual-question-answering","slug":"language-guided-visual-question-answering","title":"Language Guided Visual Question Answering: Elevate Your Multimodal Language Model Using Knowledge-Enriched Prompts","date":"2023-10-31","arxiv_id":"2310.20159","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/language-guided-visual-question-answering#ran","syntology_url":"https://syntology.ai/paper/2310.20159","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.20159"}},"official":{"repos":["declare-lab/lg-vqa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/myriad-large-multimodal-model-by-applying","slug":"myriad-large-multimodal-model-by-applying","title":"Myriad: Large Multimodal Model by Applying Vision Experts for Industrial Anomaly Detection","date":"2023-10-29","arxiv_id":"2310.19070","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":4,"n_honours":1,"n_violates":2,"n_no_contract":2,"n_pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 2 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/myriad-large-multimodal-model-by-applying#ran","syntology_url":"https://syntology.ai/paper/2310.19070","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.19070"}},"official":{"repos":["tzjtatata/myriad"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/women-wearing-lipstick-measuring-the-bias","slug":"women-wearing-lipstick-measuring-the-bias","title":"Women Wearing Lipstick: Measuring the Bias Between an Object and Its Related Gender","date":"2023-10-29","arxiv_id":"2310.19130","repositories_listed":1,"syntology":null},{"url":"/paper/apollo-zero-shot-multimodal-reasoning-with","slug":"apollo-zero-shot-multimodal-reasoning-with","title":"Apollo: Zero-shot MultiModal Reasoning with Multiple Experts","date":"2023-10-25","arxiv_id":"2310.18369","repositories_listed":1,"syntology":null},{"url":"/paper/capivara-cost-efficient-approach-for","slug":"capivara-cost-efficient-approach-for","title":"CAPIVARA: Cost-Efficient Approach for Improving Multilingual CLIP Performance on Low-Resource Languages","date":"2023-10-20","arxiv_id":"2310.13683","repositories_listed":1,"syntology":null},{"url":"/paper/icu-conquering-language-barriers-in-vision","slug":"icu-conquering-language-barriers-in-vision","title":"ICU: Conquering Language Barriers in Vision-and-Language Modeling by Dividing the Tasks into Image Captioning and Language Understanding","date":"2023-10-19","arxiv_id":"2310.12531","repositories_listed":1,"syntology":null},{"url":"/paper/rsadapter-adapting-multimodal-models-for","slug":"rsadapter-adapting-multimodal-models-for","title":"RSAdapter: Adapting Multimodal Models for Remote Sensing Visual Question Answering","date":"2023-10-19","arxiv_id":"2310.13120","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-the-fairness-of-discriminative","slug":"evaluating-the-fairness-of-discriminative","title":"Evaluating the Fairness of Discriminative Foundation Models in Computer Vision","date":"2023-10-18","arxiv_id":"2310.11867","repositories_listed":1,"syntology":null},{"url":"/paper/bounding-and-filling-a-fast-and-flexible","slug":"bounding-and-filling-a-fast-and-flexible","title":"Bounding and Filling: A Fast and Flexible Framework for Image Captioning","date":"2023-10-15","arxiv_id":"2310.09876","repositories_listed":1,"syntology":null},{"url":"/paper/from-clip-to-dino-visual-encoders-shout-in","slug":"from-clip-to-dino-visual-encoders-shout-in","title":"From CLIP to DINO: Visual Encoders Shout in Multi-modal Large Language Models","date":"2023-10-13","arxiv_id":"2310.08825","repositories_listed":1,"syntology":null},{"url":"/paper/language-models-as-knowledge-bases-for-visual","slug":"language-models-as-knowledge-bases-for-visual","title":"Language Models as Knowledge Bases for Visual Word Sense Disambiguation","date":"2023-10-03","arxiv_id":"2310.01960","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/language-models-as-knowledge-bases-for-visual#ran","syntology_url":"https://syntology.ai/paper/2310.01960","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.01960"}},"official":{"repos":["anastasiakrith/llm-for-vwsd"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mathvista-evaluating-mathematical-reasoning","slug":"mathvista-evaluating-mathematical-reasoning","title":"MathVista: Evaluating Mathematical Reasoning of Foundation Models in Visual Contexts","date":"2023-10-03","arxiv_id":"2310.02255","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mathvista-evaluating-mathematical-reasoning#ran","syntology_url":"https://syntology.ai/paper/2310.02255","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.02255"}},"official":null}},{"url":"/paper/sieve-multimodal-dataset-pruning-using-image","slug":"sieve-multimodal-dataset-pruning-using-image","title":"Sieve: Multimodal Dataset Pruning Using Image Captioning Models","date":"2023-10-03","arxiv_id":"2310.02110","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sieve-multimodal-dataset-pruning-using-image#ran","syntology_url":"https://syntology.ai/paper/2310.02110","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.02110"}},"official":{"repos":["facebookresearch/sieve"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/elip-efficient-language-image-pre-training","slug":"elip-efficient-language-image-pre-training","title":"ELIP: Efficient Language-Image Pre-training with Fewer Vision Tokens","date":"2023-09-28","arxiv_id":"2309.16738","repositories_listed":1,"syntology":null},{"url":"/paper/blip-adapter-parameter-efficient-transfer","slug":"blip-adapter-parameter-efficient-transfer","title":"BLIP-Adapter: Parameter-Efficient Transfer Learning for Mobile Screenshot Captioning","date":"2023-09-26","arxiv_id":"2309.14774","repositories_listed":1,"syntology":null},{"url":"/paper/ipic-xai-improving-pic-xai-for-enhanced-image","slug":"ipic-xai-improving-pic-xai-for-enhanced-image","title":"iPIC-XAI: Improving PIC-XAI for Enhanced Image Captioning Explanation","date":"2023-09-23","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/implicit-differentiable-outlier-detection","slug":"implicit-differentiable-outlier-detection","title":"Implicit Differentiable Outlier Detection Enable Robust Deep Multimodal Analysis","date":"2023-09-21","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/beyond-generation-harnessing-text-to-image","slug":"beyond-generation-harnessing-text-to-image","title":"Beyond Generation: Harnessing Text to Image Models for Object Detection and Segmentation","date":"2023-09-12","arxiv_id":"2309.05956","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/beyond-generation-harnessing-text-to-image#ran","syntology_url":"https://syntology.ai/paper/2309.05956","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.05956"}},"official":{"repos":["gyhandy/text2image-for-detection"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-better-multi-modal-keyphrase","slug":"towards-better-multi-modal-keyphrase","title":"Towards Better Multi-modal Keyphrase Generation via Visual Entity Enhancement and Multi-granularity Image Noise Filtering","date":"2023-09-09","arxiv_id":"2309.04734","repositories_listed":1,"syntology":null},{"url":"/paper/exchanging-based-multimodal-fusion-with","slug":"exchanging-based-multimodal-fusion-with","title":"Exchanging-based Multimodal Fusion with Transformer","date":"2023-09-05","arxiv_id":"2309.02190","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/exchanging-based-multimodal-fusion-with#ran","syntology_url":"https://syntology.ai/paper/2309.02190","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.02190"}},"official":{"repos":["recklessronan/muse"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-addressing-the-misalignment-of-object","slug":"towards-addressing-the-misalignment-of-object","title":"Towards Addressing the Misalignment of Object Proposal Evaluation for Vision-Language Tasks via Semantic Grounding","date":"2023-09-01","arxiv_id":"2309.00215","repositories_listed":1,"syntology":null},{"url":"/paper/cliptrans-transferring-visual-knowledge-with","slug":"cliptrans-transferring-visual-knowledge-with","title":"CLIPTrans: Transferring Visual Knowledge with Pre-trained Models for Multimodal Machine Translation","date":"2023-08-29","arxiv_id":"2308.15226","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/cliptrans-transferring-visual-knowledge-with#ran","syntology_url":"https://syntology.ai/paper/2308.15226","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.15226"}},"official":{"repos":["devaansh100/cliptrans"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/multicapclip-auto-encoding-prompts-for-zero","slug":"multicapclip-auto-encoding-prompts-for-zero","title":"MultiCapCLIP: Auto-Encoding Prompts for Zero-Shot Multilingual Visual Captioning","date":"2023-08-25","arxiv_id":"2308.13218","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/multicapclip-auto-encoding-prompts-for-zero#ran","syntology_url":"https://syntology.ai/paper/2308.13218","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.13218"}},"official":{"repos":["yangbang18/multicapclip"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cgt-gan-clip-guided-text-gan-for-image","slug":"cgt-gan-clip-guided-text-gan-for-image","title":"CgT-GAN: CLIP-guided Text GAN for Image Captioning","date":"2023-08-23","arxiv_id":"2308.12045","repositories_listed":1,"syntology":null},{"url":"/paper/with-a-little-help-from-your-own-past","slug":"with-a-little-help-from-your-own-past","title":"With a Little Help from your own Past: Prototypical Memory Networks for Image Captioning","date":"2023-08-23","arxiv_id":"2308.12383","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/with-a-little-help-from-your-own-past#ran","syntology_url":"https://syntology.ai/paper/2308.12383","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.12383"}},"official":{"repos":["aimagelab/pma-net"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vl-pet-vision-and-language-parameter","slug":"vl-pet-vision-and-language-parameter","title":"VL-PET: Vision-and-Language Parameter-Efficient Tuning via Granularity Control","date":"2023-08-18","arxiv_id":"2308.09804","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vl-pet-vision-and-language-parameter#ran","syntology_url":"https://syntology.ai/paper/2308.09804","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09804"}},"official":{"repos":["henryhzy/vl-pet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/visually-aware-context-modeling-for-news","slug":"visually-aware-context-modeling-for-news","title":"Visually-Aware Context Modeling for News Image Captioning","date":"2023-08-16","arxiv_id":"2308.08325","repositories_listed":1,"syntology":null},{"url":"/paper/diffusion-based-augmentation-for-captioning","slug":"diffusion-based-augmentation-for-captioning","title":"Diffusion Based Augmentation for Captioning and Retrieval in Cultural Heritage","date":"2023-08-14","arxiv_id":"2308.07151","repositories_listed":1,"syntology":null},{"url":"/paper/git-mol-a-multi-modal-large-language-model","slug":"git-mol-a-multi-modal-large-language-model","title":"GIT-Mol: A Multi-modal Large Language Model for Molecular Science with Graph, Image, and Text","date":"2023-08-14","arxiv_id":"2308.06911","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/git-mol-a-multi-modal-large-language-model#ran","syntology_url":"https://syntology.ai/paper/2308.06911","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.06911"}},"official":{"repos":["ai-hpc-research-team/git-mol"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/empowering-vision-language-models-to-follow","slug":"empowering-vision-language-models-to-follow","title":"Fine-tuning Multimodal LLMs to Follow Zero-shot Demonstrative Instructions","date":"2023-08-08","arxiv_id":"2308.04152","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":2,"n_instrument":8,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 8 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/empowering-vision-language-models-to-follow#ran","syntology_url":"https://syntology.ai/paper/2308.04152","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.04152"}},"official":{"repos":["dcdmllm/cheetah"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/ads-cap-a-framework-for-accurate-and-diverse","slug":"ads-cap-a-framework-for-accurate-and-diverse","title":"ADS-Cap: A Framework for Accurate and Diverse Stylized Captioning with Unpaired Stylistic Corpora","date":"2023-08-02","arxiv_id":"2308.01143","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-generic-enhancing-image-captioning","slug":"beyond-generic-enhancing-image-captioning","title":"Beyond Generic: Enhancing Image Captioning with Real-World Knowledge using Vision-Language Pre-Training Model","date":"2023-08-02","arxiv_id":"2308.01126","repositories_listed":1,"syntology":null},{"url":"/paper/ts-rgbd-dataset-a-novel-dataset-for-theatre","slug":"ts-rgbd-dataset-a-novel-dataset-for-theatre","title":"TS-RGBD Dataset: a Novel Dataset for Theatre Scenes Description for People with Visual Impairments","date":"2023-08-02","arxiv_id":"2308.01035","repositories_listed":1,"syntology":null},{"url":"/paper/transferable-decoding-with-visual-entities","slug":"transferable-decoding-with-visual-entities","title":"Transferable Decoding with Visual Entities for Zero-Shot Image Captioning","date":"2023-07-31","arxiv_id":"2307.16525","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":3,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"6 ran (of which 3 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/transferable-decoding-with-visual-entities#ran","syntology_url":"https://syntology.ai/paper/2307.16525","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.16525"}},"official":{"repos":["feielysia/viecap"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":3,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/exploring-annotation-free-image-captioning","slug":"exploring-annotation-free-image-captioning","title":"Exploring Annotation-free Image Captioning with Retrieval-augmented Pseudo Sentence Generation","date":"2023-07-27","arxiv_id":"2307.14750","repositories_listed":1,"syntology":null},{"url":"/paper/image-captions-are-natural-prompts-for-text","slug":"image-captions-are-natural-prompts-for-text","title":"Image Captions are Natural Prompts for Text-to-Image Models","date":"2023-07-17","arxiv_id":"2307.08526","repositories_listed":1,"syntology":null},{"url":"/paper/mblip-efficient-bootstrapping-of-multilingual","slug":"mblip-efficient-bootstrapping-of-multilingual","title":"mBLIP: Efficient Bootstrapping of Multilingual Vision-LLMs","date":"2023-07-13","arxiv_id":"2307.06930","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mblip-efficient-bootstrapping-of-multilingual#ran","syntology_url":"https://syntology.ai/paper/2307.06930","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.06930"}},"official":{"repos":["gregor-ge/mblip"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sitta-a-semantic-image-text-alignment-for","slug":"sitta-a-semantic-image-text-alignment-for","title":"Linear Alignment of Vision-language Models for Image Captioning","date":"2023-07-10","arxiv_id":"2307.05591","repositories_listed":1,"syntology":null},{"url":"/paper/clipmasterprints-fooling-contrastive-language","slug":"clipmasterprints-fooling-contrastive-language","title":"Fooling Contrastive Language-Image Pre-trained Models with CLIPMasterPrints","date":"2023-07-07","arxiv_id":"2307.03798","repositories_listed":1,"syntology":null},{"url":"/paper/journeydb-a-benchmark-for-generative-image","slug":"journeydb-a-benchmark-for-generative-image","title":"JourneyDB: A Benchmark for Generative Image Understanding","date":"2023-07-03","arxiv_id":"2307.00716","repositories_listed":1,"syntology":null},{"url":"/paper/palm-predicting-actions-through-language","slug":"palm-predicting-actions-through-language","title":"Palm: Predicting Actions through Language Models @ Ego4D Long-Term Action Anticipation Challenge 2023","date":"2023-06-28","arxiv_id":"2306.16545","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/palm-predicting-actions-through-language#ran","syntology_url":"https://syntology.ai/paper/2306.16545","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.16545"}},"official":{"repos":["dandoge/palm"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/semi-supervised-multimodal-representation","slug":"semi-supervised-multimodal-representation","title":"Semi-supervised Multimodal Representation Learning through a Global Workspace","date":"2023-06-27","arxiv_id":"2306.15711","repositories_listed":1,"syntology":null},{"url":"/paper/shikra-unleashing-multimodal-llm-s","slug":"shikra-unleashing-multimodal-llm-s","title":"Shikra: Unleashing Multimodal LLM's Referential Dialogue Magic","date":"2023-06-27","arxiv_id":"2306.15195","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/shikra-unleashing-multimodal-llm-s#ran","syntology_url":"https://syntology.ai/paper/2306.15195","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.15195"}},"official":{"repos":["shikras/shikra"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/what-makes-imagenet-look-unlike-laion","slug":"what-makes-imagenet-look-unlike-laion","title":"What Makes ImageNet Look Unlike LAION","date":"2023-06-27","arxiv_id":"2306.15769","repositories_listed":1,"syntology":null},{"url":"/paper/investigating-prompting-techniques-for-zero","slug":"investigating-prompting-techniques-for-zero","title":"Investigating Prompting Techniques for Zero- and Few-Shot Visual Question Answering","date":"2023-06-16","arxiv_id":"2306.09996","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/investigating-prompting-techniques-for-zero#ran","syntology_url":"https://syntology.ai/paper/2306.09996","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.09996"}},"official":{"repos":["rabiulcste/vqazero"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lvlm-ehub-a-comprehensive-evaluation","slug":"lvlm-ehub-a-comprehensive-evaluation","title":"LVLM-eHub: A Comprehensive Evaluation Benchmark for Large Vision-Language Models","date":"2023-06-15","arxiv_id":"2306.09265","repositories_listed":1,"syntology":null},{"url":"/paper/grounded-image-captioning-in-top-down-view","slug":"grounded-image-captioning-in-top-down-view","title":"Top-Down Framework for Weakly-supervised Grounded Image Captioning","date":"2023-06-13","arxiv_id":"2306.07490","repositories_listed":1,"syntology":null},{"url":"/paper/scalable-3d-captioning-with-pretrained-models","slug":"scalable-3d-captioning-with-pretrained-models","title":"Scalable 3D Captioning with Pretrained Models","date":"2023-06-12","arxiv_id":"2306.07279","repositories_listed":1,"syntology":null},{"url":"/paper/rewarded-soups-towards-pareto-optimal-1","slug":"rewarded-soups-towards-pareto-optimal-1","title":"Rewarded soups: towards Pareto-optimal alignment by interpolating weights fine-tuned on diverse rewards","date":"2023-06-07","arxiv_id":"2306.04488","repositories_listed":1,"syntology":null},{"url":"/paper/scicap-a-knowledge-augmented-dataset-to-study","slug":"scicap-a-knowledge-augmented-dataset-to-study","title":"SciCap+: A Knowledge Augmented Dataset to Study the Challenges of Scientific Figure Captioning","date":"2023-06-06","arxiv_id":"2306.03491","repositories_listed":1,"syntology":null},{"url":"/paper/composition-and-deformance-measuring","slug":"composition-and-deformance-measuring","title":"Composition and Deformance: Measuring Imageability with a Text-to-Image Model","date":"2023-06-05","arxiv_id":"2306.03168","repositories_listed":1,"syntology":null},{"url":"/paper/lmcap-few-shot-multilingual-image-captioning","slug":"lmcap-few-shot-multilingual-image-captioning","title":"LMCap: Few-shot Multilingual Image Captioning by Retrieval Augmented Language Model Prompting","date":"2023-05-31","arxiv_id":"2305.19821","repositories_listed":1,"syntology":null},{"url":"/paper/understanding-and-mitigating-copying-in-1","slug":"understanding-and-mitigating-copying-in-1","title":"Understanding and Mitigating Copying in Diffusion Models","date":"2023-05-31","arxiv_id":"2305.20086","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/understanding-and-mitigating-copying-in-1#ran","syntology_url":"https://syntology.ai/paper/2305.20086","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.20086"}},"official":{"repos":["somepago/dcr"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/contextual-object-detection-with-multimodal","slug":"contextual-object-detection-with-multimodal","title":"Contextual Object Detection with Multimodal Large Language Models","date":"2023-05-29","arxiv_id":"2305.18279","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/contextual-object-detection-with-multimodal#ran","syntology_url":"https://syntology.ai/paper/2305.18279","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.18279"}},"official":{"repos":["yuhangzang/contextdet"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/test-time-adaptation-with-clip-reward-for","slug":"test-time-adaptation-with-clip-reward-for","title":"Test-Time Adaptation with CLIP Reward for Zero-Shot Generalization in Vision-Language Models","date":"2023-05-29","arxiv_id":"2305.18010","repositories_listed":1,"syntology":{"n":10,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/test-time-adaptation-with-clip-reward-for#ran","syntology_url":"https://syntology.ai/paper/2305.18010","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.18010"}},"official":{"repos":["mzhaoshuai/rlcf"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/fusecap-leveraging-large-language-models-to","slug":"fusecap-leveraging-large-language-models-to","title":"FuseCap: Leveraging Large Language Models for Enriched Fused Image Captions","date":"2023-05-28","arxiv_id":"2305.17718","repositories_listed":1,"syntology":null},{"url":"/paper/crossget-cross-guided-ensemble-of-tokens-for","slug":"crossget-cross-guided-ensemble-of-tokens-for","title":"CrossGET: Cross-Guided Ensemble of Tokens for Accelerating Vision-Language Transformers","date":"2023-05-27","arxiv_id":"2305.17455","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/crossget-cross-guided-ensemble-of-tokens-for#ran","syntology_url":"https://syntology.ai/paper/2305.17455","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.17455"}},"official":{"repos":["sdc17/crossget"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/factual-a-benchmark-for-faithful-and","slug":"factual-a-benchmark-for-faithful-and","title":"FACTUAL: A Benchmark for Faithful and Consistent Textual Scene Graph Parsing","date":"2023-05-27","arxiv_id":"2305.17497","repositories_listed":1,"syntology":null},{"url":"/paper/biomedgpt-a-unified-and-generalist-biomedical","slug":"biomedgpt-a-unified-and-generalist-biomedical","title":"BiomedGPT: A Generalist Vision-Language Foundation Model for Diverse Biomedical Tasks","date":"2023-05-26","arxiv_id":"2305.17100","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/biomedgpt-a-unified-and-generalist-biomedical#ran","syntology_url":"https://syntology.ai/paper/2305.17100","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.17100"}},"official":{"repos":["taokz/biomedgpt"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/an-examination-of-the-robustness-of-reference","slug":"an-examination-of-the-robustness-of-reference","title":"An Examination of the Robustness of Reference-Free Image Captioning Evaluation Metrics","date":"2023-05-24","arxiv_id":"2305.14998","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-examination-of-the-robustness-of-reference#ran","syntology_url":"https://syntology.ai/paper/2305.14998","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14998"}},"official":{"repos":["saba96/img-cap-metrics-robustness"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cream-visually-situated-natural-language","slug":"cream-visually-situated-natural-language","title":"Visually-Situated Natural Language Understanding with Contrastive Reading Model and Frozen Large Language Models","date":"2023-05-24","arxiv_id":"2305.15080","repositories_listed":1,"syntology":null}],"record_sha256":"bda7058e2a0af31d65c8da732743a09c1721af3aa2ae92fb74a5bbd87793b924","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}