{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/image-captioning/papers/5","list_of":"/task/image-captioning","task":"Image Captioning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":5,"pages_in_order":19,"rows_per_page":100,"rows":[401,500],"of":1878,"counts":{"archive_papers_tagged":1878,"with_a_code_link":774,"where_syntology_ran_a_sample":243,"not_listed_spam_title":0,"listed":1878,"listed_where_code_ran":243,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":201,"every_run_a_failure_of_syntologys_instrument":42,"listed_with_a_run_with_no_instrument_failure":201,"listed_every_run_a_failure_of_syntologys_instrument":42,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/image-captioning","prev":"/task/image-captioning/papers/4","next":"/task/image-captioning/papers/6","papers":[{"url":"/paper/exploring-diverse-in-context-configurations","slug":"exploring-diverse-in-context-configurations","title":"Exploring Diverse In-Context Configurations for Image Captioning","date":"2023-05-24","arxiv_id":"2305.14800","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/exploring-diverse-in-context-configurations#ran","syntology_url":"https://syntology.ai/paper/2305.14800","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14800"}},"official":{"repos":["yongliang-wu/explorecfg"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/gender-biases-in-automatic-evaluation-metrics","slug":"gender-biases-in-automatic-evaluation-metrics","title":"Gender Biases in Automatic Evaluation Metrics for Image Captioning","date":"2023-05-24","arxiv_id":"2305.14711","repositories_listed":1,"syntology":null},{"url":"/paper/text-encoders-are-performance-bottlenecks-in","slug":"text-encoders-are-performance-bottlenecks-in","title":"Text encoders bottleneck compositionality in contrastive vision-language models","date":"2023-05-24","arxiv_id":"2305.14897","repositories_listed":1,"syntology":null},{"url":"/paper/memecap-a-dataset-for-captioning-and","slug":"memecap-a-dataset-for-captioning-and","title":"MemeCap: A Dataset for Captioning and Interpreting Memes","date":"2023-05-23","arxiv_id":"2305.13703","repositories_listed":1,"syntology":null},{"url":"/paper/pic-xai-post-hoc-image-captioning-explanation","slug":"pic-xai-post-hoc-image-captioning-explanation","title":"PIC-XAI: Post-hoc Image Captioning Explanation using Segmentation","date":"2023-05-23","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/what-makes-for-good-visual-tokenizers-for","slug":"what-makes-for-good-visual-tokenizers-for","title":"What Makes for Good Visual Tokenizers for Large Language Models?","date":"2023-05-20","arxiv_id":"2305.12223","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/what-makes-for-good-visual-tokenizers-for#ran","syntology_url":"https://syntology.ai/paper/2305.12223","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.12223"}},"official":{"repos":["tencentarc/gvt"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/brain-captioning-decoding-human-brain","slug":"brain-captioning-decoding-human-brain","title":"Brain Captioning: Decoding human brain activity into images and text","date":"2023-05-19","arxiv_id":"2305.11560","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-vision-language-pre-training-with","slug":"enhancing-vision-language-pre-training-with","title":"Enhancing Vision-Language Pre-Training with Jointly Learned Questioner and Dense Captioner","date":"2023-05-19","arxiv_id":"2305.11769","repositories_listed":1,"syntology":null},{"url":"/paper/bridging-the-domain-gap-self-supervised-3d","slug":"bridging-the-domain-gap-self-supervised-3d","title":"Bridging the Domain Gap: Self-Supervised 3D Scene Understanding with Foundation Models","date":"2023-05-15","arxiv_id":"2305.08776","repositories_listed":1,"syntology":null},{"url":"/paper/imaginator-pre-trained-image-text-joint","slug":"imaginator-pre-trained-image-text-joint","title":"IMAGINATOR: Pre-Trained Image+Text Joint Embeddings using Word-Level Grounding of Images","date":"2023-05-12","arxiv_id":"2305.10438","repositories_listed":1,"syntology":null},{"url":"/paper/infometic-an-informative-metric-for-reference","slug":"infometic-an-informative-metric-for-reference","title":"InfoMetIC: An Informative Metric for Reference-free Image Caption Evaluation","date":"2023-05-10","arxiv_id":"2305.06002","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/infometic-an-informative-metric-for-reference#ran","syntology_url":"https://syntology.ai/paper/2305.06002","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.06002"}},"official":{"repos":["hawlyq/infometic"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-suite-of-generative-tasks-for-multi-level","slug":"a-suite-of-generative-tasks-for-multi-level","title":"A Suite of Generative Tasks for Multi-Level Multimodal Webpage Understanding","date":"2023-05-05","arxiv_id":"2305.03668","repositories_listed":1,"syntology":null},{"url":"/paper/data-curation-for-image-captioning-with-text","slug":"data-curation-for-image-captioning-with-text","title":"The Role of Data Curation in Image Captioning","date":"2023-05-05","arxiv_id":"2305.03610","repositories_listed":1,"syntology":null},{"url":"/paper/caption-anything-interactive-image","slug":"caption-anything-interactive-image","title":"Caption Anything: Interactive Image Description with Diverse Multimodal Controls","date":"2023-05-04","arxiv_id":"2305.02677","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/caption-anything-interactive-image#ran","syntology_url":"https://syntology.ai/paper/2305.02677","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.02677"}},"official":{"repos":["ttengwang/caption-anything"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/multimodal-data-augmentation-for-image","slug":"multimodal-data-augmentation-for-image","title":"Multimodal Data Augmentation for Image Captioning using Diffusion Models","date":"2023-05-03","arxiv_id":"2305.01855","repositories_listed":1,"syntology":null},{"url":"/paper/transforming-visual-scene-graphs-to-image","slug":"transforming-visual-scene-graphs-to-image","title":"Transforming Visual Scene Graphs to Image Captions","date":"2023-05-03","arxiv_id":"2305.02177","repositories_listed":1,"syntology":null},{"url":"/paper/from-association-to-generation-text-only","slug":"from-association-to-generation-text-only","title":"From Association to Generation: Text-only Captioning by Unsupervised Cross-modal Mapping","date":"2023-04-26","arxiv_id":"2304.13273","repositories_listed":1,"syntology":null},{"url":"/paper/ttida-controllable-generative-data","slug":"ttida-controllable-generative-data","title":"TTIDA: Controllable Generative Data Augmentation via Text-to-Text and Text-to-Image Models","date":"2023-04-18","arxiv_id":"2304.08821","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ttida-controllable-generative-data#ran","syntology_url":"https://syntology.ai/paper/2304.08821","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.08821"}},"official":{"repos":["yuweiyin/ttida"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/valor-vision-audio-language-omni-perception","slug":"valor-vision-audio-language-omni-perception","title":"VALOR: Vision-Audio-Language Omni-Perception Pretraining Model and Dataset","date":"2023-04-17","arxiv_id":"2304.08345","repositories_listed":1,"syntology":null},{"url":"/paper/model-agnostic-gender-debiased-image","slug":"model-agnostic-gender-debiased-image","title":"Model-Agnostic Gender Debiased Image Captioning","date":"2023-04-07","arxiv_id":"2304.03693","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/model-agnostic-gender-debiased-image#ran","syntology_url":"https://syntology.ai/paper/2304.03693","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.03693"}},"official":{"repos":["rebnej/libra"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/uncurated-image-text-datasets-shedding-light","slug":"uncurated-image-text-datasets-shedding-light","title":"Uncurated Image-Text Datasets: Shedding Light on Demographic Bias","date":"2023-04-06","arxiv_id":"2304.02828","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/uncurated-image-text-datasets-shedding-light#ran","syntology_url":"https://syntology.ai/paper/2304.02828","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.02828"}},"official":{"repos":["noagarcia/phase"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/cross-domain-image-captioning-with","slug":"cross-domain-image-captioning-with","title":"Cross-Domain Image Captioning with Discriminative Finetuning","date":"2023-04-04","arxiv_id":"2304.01662","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cross-domain-image-captioning-with#ran","syntology_url":"https://syntology.ai/paper/2304.01662","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.01662"}},"official":{"repos":["facebookresearch/EGG"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/grand-challenge-on-detecting-cheapfakes","slug":"grand-challenge-on-detecting-cheapfakes","title":"Grand Challenge On Detecting Cheapfakes","date":"2023-04-03","arxiv_id":"2304.01328","repositories_listed":1,"syntology":null},{"url":"/paper/autoad-movie-description-in-context","slug":"autoad-movie-description-in-context","title":"AutoAD: Movie Description in Context","date":"2023-03-29","arxiv_id":"2303.16899","repositories_listed":1,"syntology":{"n":13,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/autoad-movie-description-in-context#ran","syntology_url":"https://syntology.ai/paper/2303.16899","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.16899"}},"official":{"repos":["Soldelli/MAD"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/multimodal-image-text-matching-improves","slug":"multimodal-image-text-matching-improves","title":"Multimodal Image-Text Matching Improves Retrieval-based Chest X-Ray Report Generation","date":"2023-03-29","arxiv_id":"2303.17579","repositories_listed":1,"syntology":null},{"url":"/paper/magvlt-masked-generative-vision-and-language","slug":"magvlt-masked-generative-vision-and-language","title":"MAGVLT: Masked Generative Vision-and-Language Transformer","date":"2023-03-21","arxiv_id":"2303.12208","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":5,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 5 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/magvlt-masked-generative-vision-and-language#ran","syntology_url":"https://syntology.ai/paper/2303.12208","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.12208"}},"official":{"repos":["kakaobrain/magvlt"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":5,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/positive-augmented-constrastive-learning-for","slug":"positive-augmented-constrastive-learning-for","title":"Positive-Augmented Contrastive Learning for Image and Video Captioning Evaluation","date":"2023-03-21","arxiv_id":"2303.12112","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":2,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/positive-augmented-constrastive-learning-for#ran","syntology_url":"https://syntology.ai/paper/2303.12112","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.12112"}},"official":{"repos":["aimagelab/pacscore"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/chatgpt-asks-blip-2-answers-automatic","slug":"chatgpt-asks-blip-2-answers-automatic","title":"ChatGPT Asks, BLIP-2 Answers: Automatic Questioning Towards Enriched Visual Descriptions","date":"2023-03-12","arxiv_id":"2303.06594","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/chatgpt-asks-blip-2-answers-automatic#ran","syntology_url":"https://syntology.ai/paper/2303.06594","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.06594"}},"official":{"repos":["vision-cair/chatcaptioner"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/contrastive-language-image-pretrained-clip","slug":"contrastive-language-image-pretrained-clip","title":"Adapting Contrastive Language-Image Pretrained (CLIP) Models for Out-of-Distribution Detection","date":"2023-03-10","arxiv_id":"2303.05828","repositories_listed":1,"syntology":null},{"url":"/paper/decap-decoding-clip-latents-for-zero-shot","slug":"decap-decoding-clip-latents-for-zero-shot","title":"DeCap: Decoding CLIP Latents for Zero-Shot Captioning via Text-Only Training","date":"2023-03-06","arxiv_id":"2303.03032","repositories_listed":1,"syntology":null},{"url":"/paper/neighborhood-contrastive-transformer-for","slug":"neighborhood-contrastive-transformer-for","title":"Neighborhood Contrastive Transformer for Change Captioning","date":"2023-03-06","arxiv_id":"2303.03171","repositories_listed":1,"syntology":null},{"url":"/paper/conzic-controllable-zero-shot-image","slug":"conzic-controllable-zero-shot-image","title":"ConZIC: Controllable Zero-shot Image Captioning by Sampling-Based Polishing","date":"2023-03-04","arxiv_id":"2303.02437","repositories_listed":1,"syntology":{"n":10,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":7,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/conzic-controllable-zero-shot-image#ran","syntology_url":"https://syntology.ai/paper/2303.02437","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.02437"}},"official":{"repos":["joeyz0z/conzic"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/fame-vil-multi-tasking-vision-language-model","slug":"fame-vil-multi-tasking-vision-language-model","title":"FAME-ViL: Multi-Tasking Vision-Language Model for Heterogeneous Fashion Tasks","date":"2023-03-04","arxiv_id":"2303.02483","repositories_listed":1,"syntology":null},{"url":"/paper/contextual-net-a-multimodal-vision-language","slug":"contextual-net-a-multimodal-vision-language","title":"ConTEXTual Net: A Multimodal Vision-Language Model for Segmentation of Pneumothorax","date":"2023-03-02","arxiv_id":"2303.01615","repositories_listed":1,"syntology":null},{"url":"/paper/language-is-not-all-you-need-aligning-1","slug":"language-is-not-all-you-need-aligning-1","title":"Language Is Not All You Need: Aligning Perception with Language Models","date":"2023-02-27","arxiv_id":"2302.14045","repositories_listed":1,"syntology":null},{"url":"/paper/retrieval-augmented-image-captioning","slug":"retrieval-augmented-image-captioning","title":"Retrieval-augmented Image Captioning","date":"2023-02-16","arxiv_id":"2302.08268","repositories_listed":1,"syntology":null},{"url":"/paper/tuning-computer-vision-models-with-task","slug":"tuning-computer-vision-models-with-task","title":"Tuning computer vision models with task rewards","date":"2023-02-16","arxiv_id":"2302.08242","repositories_listed":1,"syntology":null},{"url":"/paper/towards-local-visual-modeling-for-image","slug":"towards-local-visual-modeling-for-image","title":"Towards Local Visual Modeling for Image Captioning","date":"2023-02-13","arxiv_id":"2302.06098","repositories_listed":1,"syntology":null},{"url":"/paper/transform-contrast-and-tell-coherent-entity","slug":"transform-contrast-and-tell-coherent-entity","title":"Transform, Contrast and Tell: Coherent Entity-Aware Multi-Image Captioning","date":"2023-02-04","arxiv_id":"2302.02124","repositories_listed":1,"syntology":null},{"url":"/paper/ic-3-image-captioning-by-committee-consensus","slug":"ic-3-image-captioning-by-committee-consensus","title":"IC3: Image Captioning by Committee Consensus","date":"2023-02-02","arxiv_id":"2302.01328","repositories_listed":1,"syntology":null},{"url":"/paper/paraphrase-acquisition-from-image-captions","slug":"paraphrase-acquisition-from-image-captions","title":"Paraphrase Acquisition from Image Captions","date":"2023-01-26","arxiv_id":"2301.11030","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-the-synergy-between-vision-language","slug":"exploring-the-synergy-between-vision-language","title":"Exploring the Synergy Between Vision-Language Pretraining and ChatGPT for Artwork Captioning: A Preliminary Study","date":"2023-01-21","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/visual-semantic-relatedness-dataset-for-image","slug":"visual-semantic-relatedness-dataset-for-image","title":"Visual Semantic Relatedness Dataset for Image Captioning","date":"2023-01-20","arxiv_id":"2301.08784","repositories_listed":1,"syntology":null},{"url":"/paper/see-think-confirm-interactive-prompting","slug":"see-think-confirm-interactive-prompting","title":"See, Think, Confirm: Interactive Prompting Between Vision and Language Models for Knowledge-based Visual Reasoning","date":"2023-01-12","arxiv_id":"2301.05226","repositories_listed":1,"syntology":null},{"url":"/paper/adaptively-clustering-neighbor-elements-for","slug":"adaptively-clustering-neighbor-elements-for","title":"Adaptively Clustering Neighbor Elements for Image-Text Generation","date":"2023-01-05","arxiv_id":"2301.01955","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-interpretability-of-attention-networks","slug":"on-the-interpretability-of-attention-networks","title":"On the Interpretability of Attention Networks","date":"2022-12-30","arxiv_id":"2212.14776","repositories_listed":1,"syntology":null},{"url":"/paper/noise-aware-learning-from-web-crawled-image","slug":"noise-aware-learning-from-web-crawled-image","title":"Noise-aware Learning from Web-crawled Image-Text Data for Image Captioning","date":"2022-12-27","arxiv_id":"2212.13563","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/noise-aware-learning-from-web-crawled-image#ran","syntology_url":"https://syntology.ai/paper/2212.13563","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.13563"}},"official":{"repos":["kakaobrain/noc"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/on-realization-of-intelligent-decision-making","slug":"on-realization-of-intelligent-decision-making","title":"On Realization of Intelligent Decision-Making in the Real World: A Foundation Decision Model Perspective","date":"2022-12-24","arxiv_id":"2212.12669","repositories_listed":1,"syntology":null},{"url":"/paper/position-guided-text-prompt-for-vision","slug":"position-guided-text-prompt-for-vision","title":"Position-guided Text Prompt for Vision-Language Pre-training","date":"2022-12-19","arxiv_id":"2212.09737","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/position-guided-text-prompt-for-vision#ran","syntology_url":"https://syntology.ai/paper/2212.09737","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.09737"}},"official":{"repos":["sail-sg/ptp"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/transferring-general-multimodal-pretrained","slug":"transferring-general-multimodal-pretrained","title":"Transferring General Multimodal Pretrained Models to Text Recognition","date":"2022-12-19","arxiv_id":"2212.09297","repositories_listed":1,"syntology":null},{"url":"/paper/are-multimodal-models-robust-to-image-and","slug":"are-multimodal-models-robust-to-image-and","title":"Benchmarking Robustness of Multimodal Image-Text Models under Distribution Shift","date":"2022-12-15","arxiv_id":"2212.08044","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/are-multimodal-models-robust-to-image-and#ran","syntology_url":"https://syntology.ai/paper/2212.08044","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.08044"}},"official":null}},{"url":"/paper/reveal-retrieval-augmented-visual-language","slug":"reveal-retrieval-augmented-visual-language","title":"REVEAL: Retrieval-Augmented Visual-Language Pre-Training with Multi-Source Multimodal Knowledge Memory","date":"2022-12-10","arxiv_id":"2212.05221","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-testing-of-computer-vision-models","slug":"adaptive-testing-of-computer-vision-models","title":"Adaptive Testing of Computer Vision Models","date":"2022-12-06","arxiv_id":"2212.02774","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/adaptive-testing-of-computer-vision-models#ran","syntology_url":"https://syntology.ai/paper/2212.02774","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.02774"}},"official":{"repos":["i-gao/adavision"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/switching-to-discriminative-image-captioning","slug":"switching-to-discriminative-image-captioning","title":"Switching to Discriminative Image Captioning by Relieving a Bottleneck of Reinforcement Learning","date":"2022-12-06","arxiv_id":"2212.03230","repositories_listed":1,"syntology":null},{"url":"/paper/clid-controlled-length-image-descriptions","slug":"clid-controlled-length-image-descriptions","title":"CLID: Controlled-Length Image Descriptions with Limited Data","date":"2022-11-27","arxiv_id":"2211.14835","repositories_listed":1,"syntology":null},{"url":"/paper/aesthetically-relevant-image-captioning","slug":"aesthetically-relevant-image-captioning","title":"Aesthetically Relevant Image Captioning","date":"2022-11-25","arxiv_id":"2211.15378","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-discrete-diffusion-models-for-image","slug":"exploring-discrete-diffusion-models-for-image","title":"Exploring Discrete Diffusion Models for Image Captioning","date":"2022-11-21","arxiv_id":"2211.11694","repositories_listed":1,"syntology":{"n":18,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":4,"n_honours":2,"n_violates":0,"n_no_contract":12,"n_pointer_only":5,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 2 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/exploring-discrete-diffusion-models-for-image#ran","syntology_url":"https://syntology.ai/paper/2211.11694","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.11694"}},"official":{"repos":["buxiangzhiren/ddcap"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/i-can-t-believe-there-s-no-images-learning","slug":"i-can-t-believe-there-s-no-images-learning","title":"I Can't Believe There's No Images! Learning Visual Tasks Using only Language Supervision","date":"2022-11-17","arxiv_id":"2211.09778","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/i-can-t-believe-there-s-no-images-learning#ran","syntology_url":"https://syntology.ai/paper/2211.09778","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.09778"}},"official":{"repos":["allenai/close"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/progressive-tree-structured-prototype-network","slug":"progressive-tree-structured-prototype-network","title":"Progressive Tree-Structured Prototype Network for End-to-End Image Captioning","date":"2022-11-17","arxiv_id":"2211.09460","repositories_listed":1,"syntology":null},{"url":"/paper/promptcap-prompt-guided-task-aware-image","slug":"promptcap-prompt-guided-task-aware-image","title":"PromptCap: Prompt-Guided Task-Aware Image Captioning","date":"2022-11-15","arxiv_id":"2211.09699","repositories_listed":1,"syntology":null},{"url":"/paper/large-scale-bidirectional-training-for-zero","slug":"large-scale-bidirectional-training-for-zero","title":"Large-Scale Bidirectional Training for Zero-Shot Image Captioning","date":"2022-11-13","arxiv_id":"2211.06774","repositories_listed":1,"syntology":null},{"url":"/paper/deltanet-conditional-medical-report-1","slug":"deltanet-conditional-medical-report-1","title":"DeltaNet:Conditional Medical Report Generation for COVID-19 Diagnosis","date":"2022-11-12","arxiv_id":"2211.13229","repositories_listed":1,"syntology":null},{"url":"/paper/rsvg-exploring-data-and-models-for-visual","slug":"rsvg-exploring-data-and-models-for-visual","title":"RSVG: Exploring Data and Models for Visual Grounding on Remote Sensing Data","date":"2022-10-23","arxiv_id":"2210.12634","repositories_listed":1,"syntology":null},{"url":"/paper/posescript-3d-human-poses-from-natural","slug":"posescript-3d-human-poses-from-natural","title":"PoseScript: Linking 3D Human Poses and Natural Language","date":"2022-10-21","arxiv_id":"2210.11795","repositories_listed":1,"syntology":null},{"url":"/paper/visual-spatial-description-controlled-spatial","slug":"visual-spatial-description-controlled-spatial","title":"Visual Spatial Description: Controlled Spatial-Oriented Image-to-Text Generation","date":"2022-10-20","arxiv_id":"2210.11109","repositories_listed":1,"syntology":null},{"url":"/paper/vision-language-pre-training-basics-recent","slug":"vision-language-pre-training-basics-recent","title":"Vision-Language Pre-training: Basics, Recent Advances, and Future Trends","date":"2022-10-17","arxiv_id":"2210.09263","repositories_listed":1,"syntology":null},{"url":"/paper/mapl-parameter-efficient-adaptation-of","slug":"mapl-parameter-efficient-adaptation-of","title":"MAPL: Parameter-Efficient Adaptation of Unimodal Pre-Trained Models for Vision-Language Few-Shot Prompting","date":"2022-10-13","arxiv_id":"2210.07179","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mapl-parameter-efficient-adaptation-of#ran","syntology_url":"https://syntology.ai/paper/2210.07179","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.07179"}},"official":{"repos":["mair-lab/mapl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-language-maps-for-robot-navigation","slug":"visual-language-maps-for-robot-navigation","title":"Visual Language Maps for Robot Navigation","date":"2022-10-11","arxiv_id":"2210.05714","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/visual-language-maps-for-robot-navigation#ran","syntology_url":"https://syntology.ai/paper/2210.05714","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.05714"}},"official":null}},{"url":"/paper/clip-diffusion-lm-apply-diffusion-model-on","slug":"clip-diffusion-lm-apply-diffusion-model-on","title":"CLIP-Diffusion-LM: Apply Diffusion Model on Image Captioning","date":"2022-10-10","arxiv_id":"2210.04559","repositories_listed":1,"syntology":null},{"url":"/paper/mmt-image-guided-story-ending-generation-with","slug":"mmt-image-guided-story-ending-generation-with","title":"MMT: Image-guided Story Ending Generation with Multimodal Memory Transformer","date":"2022-10-10","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/not-all-errors-are-equal-learning-text","slug":"not-all-errors-are-equal-learning-text","title":"Not All Errors are Equal: Learning Text Generation Metrics using Stratified Error Synthesis","date":"2022-10-10","arxiv_id":"2210.05035","repositories_listed":1,"syntology":null},{"url":"/paper/open-vocabulary-semantic-segmentation-with","slug":"open-vocabulary-semantic-segmentation-with","title":"Open-Vocabulary Semantic Segmentation with Mask-adapted CLIP","date":"2022-10-09","arxiv_id":"2210.04150","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/open-vocabulary-semantic-segmentation-with#ran","syntology_url":"https://syntology.ai/paper/2210.04150","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.04150"}},"official":{"repos":["facebookresearch/ov-seg"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/towards-multi-modal-sarcasm-detection-via","slug":"towards-multi-modal-sarcasm-detection-via","title":"Towards Multi-Modal Sarcasm Detection via Hierarchical Congruity Modeling with Knowledge Enhancement","date":"2022-10-07","arxiv_id":"2210.03501","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/towards-multi-modal-sarcasm-detection-via#ran","syntology_url":"https://syntology.ai/paper/2210.03501","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.03501"}},"official":{"repos":["less-and-less-bugs/hkemodel"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-collocate-visual-linguistic","slug":"learning-to-collocate-visual-linguistic","title":"Learning to Collocate Visual-Linguistic Neural Modules for Image Captioning","date":"2022-10-04","arxiv_id":"2210.01338","repositories_listed":1,"syntology":null},{"url":"/paper/smallcap-lightweight-image-captioning","slug":"smallcap-lightweight-image-captioning","title":"SmallCap: Lightweight Image Captioning Prompted with Retrieval Augmentation","date":"2022-09-30","arxiv_id":"2209.15323","repositories_listed":1,"syntology":null},{"url":"/paper/mr-right-multimodal-retrieval-on","slug":"mr-right-multimodal-retrieval-on","title":"Mr. Right: Multimodal Retrieval on Representation of ImaGe witH Text","date":"2022-09-28","arxiv_id":"2209.13764","repositories_listed":1,"syntology":null},{"url":"/paper/learning-distinct-and-representative-modes","slug":"learning-distinct-and-representative-modes","title":"Learning Distinct and Representative Styles for Image Captioning","date":"2022-09-17","arxiv_id":"2209.08231","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/learning-distinct-and-representative-modes#ran","syntology_url":"https://syntology.ai/paper/2209.08231","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.08231"}},"official":{"repos":["bladewaltz1/modecap"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/belief-revision-based-caption-re-ranker-with","slug":"belief-revision-based-caption-re-ranker-with","title":"Belief Revision based Caption Re-ranker with Visual Semantic Information","date":"2022-09-16","arxiv_id":"2209.08163","repositories_listed":1,"syntology":null},{"url":"/paper/lavis-a-library-for-language-vision","slug":"lavis-a-library-for-language-vision","title":"LAVIS: A Library for Language-Vision Intelligence","date":"2022-09-15","arxiv_id":"2209.09019","repositories_listed":1,"syntology":null},{"url":"/paper/m-4i-multi-modal-models-membership-inference","slug":"m-4i-multi-modal-models-membership-inference","title":"M^4I: Multi-modal Models Membership Inference","date":"2022-09-15","arxiv_id":"2209.06997","repositories_listed":1,"syntology":null},{"url":"/paper/pali-a-jointly-scaled-multilingual-language","slug":"pali-a-jointly-scaled-multilingual-language","title":"PaLI: A Jointly-Scaled Multilingual Language-Image Model","date":"2022-09-14","arxiv_id":"2209.06794","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/pali-a-jointly-scaled-multilingual-language#ran","syntology_url":"https://syntology.ai/paper/2209.06794","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.06794"}},"official":{"repos":["google-research/big_vision"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/aspect-based-sentiment-classification-with-1","slug":"aspect-based-sentiment-classification-with-1","title":"Target-oriented Sentiment Classification with Sequential Cross-modal Semantic Graph","date":"2022-08-19","arxiv_id":"2208.09417","repositories_listed":1,"syntology":null},{"url":"/paper/gsrformer-grounded-situation-recognition","slug":"gsrformer-grounded-situation-recognition","title":"GSRFormer: Grounded Situation Recognition Transformer with Alternate Semantic Attention Refinement","date":"2022-08-18","arxiv_id":"2208.08965","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":4,"n_honours":2,"n_violates":0,"n_no_contract":3,"n_pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/gsrformer-grounded-situation-recognition#ran","syntology_url":"https://syntology.ai/paper/2208.08965","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.08965"}},"official":{"repos":["zhiqic/gsrformer"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/vault-augmenting-the-vision-and-language","slug":"vault-augmenting-the-vision-and-language","title":"VAuLT: Augmenting the Vision-and-Language Transformer for Sentiment Classification on Social Media","date":"2022-08-18","arxiv_id":"2208.09021","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vault-augmenting-the-vision-and-language#ran","syntology_url":"https://syntology.ai/paper/2208.09021","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.09021"}},"official":{"repos":["gchochla/vault"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/illume-rationalizing-vision-language-models","slug":"illume-rationalizing-vision-language-models","title":"ILLUME: Rationalizing Vision-Language Models through Human Interactions","date":"2022-08-17","arxiv_id":"2208.08241","repositories_listed":1,"syntology":null},{"url":"/paper/expansionnet-v2-block-static-expansion-in","slug":"expansionnet-v2-block-static-expansion-in","title":"Exploiting Multiple Sequence Lengths in Fast End to End Training for Image Captioning","date":"2022-08-13","arxiv_id":"2208.06551","repositories_listed":1,"syntology":null},{"url":"/paper/prompt-tuning-for-generative-multimodal","slug":"prompt-tuning-for-generative-multimodal","title":"Prompt Tuning for Generative Multimodal Pretrained Models","date":"2022-08-04","arxiv_id":"2208.02532","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/prompt-tuning-for-generative-multimodal#ran","syntology_url":"https://syntology.ai/paper/2208.02532","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.02532"}},"official":{"repos":["ofa-sys/ofa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-modeling-of-future-context-for","slug":"efficient-modeling-of-future-context-for","title":"Efficient Modeling of Future Context for Image Captioning","date":"2022-07-22","arxiv_id":"2207.10897","repositories_listed":1,"syntology":null},{"url":"/paper/rethinking-the-reference-based-distinctive","slug":"rethinking-the-reference-based-distinctive","title":"Rethinking the Reference-based Distinctive Image Captioning","date":"2022-07-22","arxiv_id":"2207.11118","repositories_listed":1,"syntology":null},{"url":"/paper/zero-shot-video-captioning-with-evolving","slug":"zero-shot-video-captioning-with-evolving","title":"Zero-Shot Video Captioning with Evolving Pseudo-Tokens","date":"2022-07-22","arxiv_id":"2207.11100","repositories_listed":1,"syntology":null},{"url":"/paper/dual-branch-hybrid-learning-network-for","slug":"dual-branch-hybrid-learning-network-for","title":"Dual-branch Hybrid Learning Network for Unbiased Scene Graph Generation","date":"2022-07-16","arxiv_id":"2207.07913","repositories_listed":1,"syntology":null},{"url":"/paper/linecap-line-charts-for-data-visualization","slug":"linecap-line-charts-for-data-visualization","title":"LineCap: Line Charts for Data Visualization Captioning Models","date":"2022-07-15","arxiv_id":"2207.07243","repositories_listed":1,"syntology":null},{"url":"/paper/detecting-and-recovering-sequential-deepfake","slug":"detecting-and-recovering-sequential-deepfake","title":"Detecting and Recovering Sequential DeepFake Manipulation","date":"2022-07-05","arxiv_id":"2207.02204","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":2,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/detecting-and-recovering-sequential-deepfake#ran","syntology_url":"https://syntology.ai/paper/2207.02204","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.02204"}},"official":{"repos":["rshaojimmy/seqdeepfake"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/milanlp-at-semeval-2022-task-5-using","slug":"milanlp-at-semeval-2022-task-5-using","title":"MilaNLP at SemEval-2022 Task 5: Using Perceiver IO for Detecting Misogynous Memes with Text and Image Modalities","date":"2022-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/zodiac-zoneout-dropout-injection-attention","slug":"zodiac-zoneout-dropout-injection-attention","title":"ZoDIAC: Zoneout Dropout Injection Attention Calculation","date":"2022-06-28","arxiv_id":"2206.14263","repositories_listed":1,"syntology":null},{"url":"/paper/what-is-where-by-looking-weakly-supervised","slug":"what-is-where-by-looking-weakly-supervised","title":"What is Where by Looking: Weakly-Supervised Open-World Phrase-Grounding without Text Inputs","date":"2022-06-19","arxiv_id":"2206.09358","repositories_listed":1,"syntology":null},{"url":"/paper/a-unified-sequence-interface-for-vision-tasks","slug":"a-unified-sequence-interface-for-vision-tasks","title":"A Unified Sequence Interface for Vision Tasks","date":"2022-06-15","arxiv_id":"2206.07669","repositories_listed":1,"syntology":null},{"url":"/paper/coarse-to-fine-vision-language-pre-training","slug":"coarse-to-fine-vision-language-pre-training","title":"Coarse-to-Fine Vision-Language Pre-training with Fusion in the Backbone","date":"2022-06-15","arxiv_id":"2206.07643","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/coarse-to-fine-vision-language-pre-training#ran","syntology_url":"https://syntology.ai/paper/2206.07643","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.07643"}},"official":{"repos":["microsoft/fiber"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/comprehending-and-ordering-semantics-for-1","slug":"comprehending-and-ordering-semantics-for-1","title":"Comprehending and Ordering Semantics for Image Captioning","date":"2022-06-14","arxiv_id":"2206.06930","repositories_listed":1,"syntology":null},{"url":"/paper/language-models-are-general-purpose","slug":"language-models-are-general-purpose","title":"Language Models are General-Purpose Interfaces","date":"2022-06-13","arxiv_id":"2206.06336","repositories_listed":1,"syntology":null}],"record_sha256":"9f8daa41a2439112b89a77b5e5cc8fc4b403e2e82bcaac03c776fee0a60da766","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}