{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/image-captioning/papers/ran/2","list_of":"/task/image-captioning","task":"Image Captioning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":2,"pages_in_order":3,"rows_per_page":100,"rows":[101,200],"of":243,"counts":{"archive_papers_tagged":1878,"with_a_code_link":774,"where_syntology_ran_a_sample":243,"not_listed_spam_title":0,"listed":1878,"listed_where_code_ran":243,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":201,"every_run_a_failure_of_syntologys_instrument":42,"listed_with_a_run_with_no_instrument_failure":201,"listed_every_run_a_failure_of_syntologys_instrument":42,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/image-captioning/papers/ran/1","prev":"/task/image-captioning/papers/ran/1","next":"/task/image-captioning/papers/ran/3","papers":[{"url":"/paper/an-examination-of-the-robustness-of-reference","slug":"an-examination-of-the-robustness-of-reference","title":"An Examination of the Robustness of Reference-Free Image Captioning Evaluation Metrics","date":"2023-05-24","arxiv_id":"2305.14998","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-examination-of-the-robustness-of-reference#ran","syntology_url":"https://syntology.ai/paper/2305.14998","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14998"}},"official":{"repos":["saba96/img-cap-metrics-robustness"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/what-makes-for-good-visual-tokenizers-for","slug":"what-makes-for-good-visual-tokenizers-for","title":"What Makes for Good Visual Tokenizers for Large Language Models?","date":"2023-05-20","arxiv_id":"2305.12223","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/what-makes-for-good-visual-tokenizers-for#ran","syntology_url":"https://syntology.ai/paper/2305.12223","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.12223"}},"official":{"repos":["tencentarc/gvt"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/infometic-an-informative-metric-for-reference","slug":"infometic-an-informative-metric-for-reference","title":"InfoMetIC: An Informative Metric for Reference-free Image Caption Evaluation","date":"2023-05-10","arxiv_id":"2305.06002","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/infometic-an-informative-metric-for-reference#ran","syntology_url":"https://syntology.ai/paper/2305.06002","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.06002"}},"official":{"repos":["hawlyq/infometic"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/caption-anything-interactive-image","slug":"caption-anything-interactive-image","title":"Caption Anything: Interactive Image Description with Diverse Multimodal Controls","date":"2023-05-04","arxiv_id":"2305.02677","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/caption-anything-interactive-image#ran","syntology_url":"https://syntology.ai/paper/2305.02677","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.02677"}},"official":{"repos":["ttengwang/caption-anything"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/ttida-controllable-generative-data","slug":"ttida-controllable-generative-data","title":"TTIDA: Controllable Generative Data Augmentation via Text-to-Text and Text-to-Image Models","date":"2023-04-18","arxiv_id":"2304.08821","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ttida-controllable-generative-data#ran","syntology_url":"https://syntology.ai/paper/2304.08821","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.08821"}},"official":{"repos":["yuweiyin/ttida"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/uncurated-image-text-datasets-shedding-light","slug":"uncurated-image-text-datasets-shedding-light","title":"Uncurated Image-Text Datasets: Shedding Light on Demographic Bias","date":"2023-04-06","arxiv_id":"2304.02828","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/uncurated-image-text-datasets-shedding-light#ran","syntology_url":"https://syntology.ai/paper/2304.02828","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.02828"}},"official":{"repos":["noagarcia/phase"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/cross-domain-image-captioning-with","slug":"cross-domain-image-captioning-with","title":"Cross-Domain Image Captioning with Discriminative Finetuning","date":"2023-04-04","arxiv_id":"2304.01662","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cross-domain-image-captioning-with#ran","syntology_url":"https://syntology.ai/paper/2304.01662","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.01662"}},"official":{"repos":["facebookresearch/EGG"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/autoad-movie-description-in-context","slug":"autoad-movie-description-in-context","title":"AutoAD: Movie Description in Context","date":"2023-03-29","arxiv_id":"2303.16899","repositories_listed":1,"syntology":{"n":13,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/autoad-movie-description-in-context#ran","syntology_url":"https://syntology.ai/paper/2303.16899","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.16899"}},"official":{"repos":["Soldelli/MAD"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/positive-augmented-constrastive-learning-for","slug":"positive-augmented-constrastive-learning-for","title":"Positive-Augmented Contrastive Learning for Image and Video Captioning Evaluation","date":"2023-03-21","arxiv_id":"2303.12112","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":2,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/positive-augmented-constrastive-learning-for#ran","syntology_url":"https://syntology.ai/paper/2303.12112","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.12112"}},"official":{"repos":["aimagelab/pacscore"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/magvlt-masked-generative-vision-and-language","slug":"magvlt-masked-generative-vision-and-language","title":"MAGVLT: Masked Generative Vision-and-Language Transformer","date":"2023-03-21","arxiv_id":"2303.12208","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":5,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 5 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/magvlt-masked-generative-vision-and-language#ran","syntology_url":"https://syntology.ai/paper/2303.12208","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.12208"}},"official":{"repos":["kakaobrain/magvlt"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":5,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/chatgpt-asks-blip-2-answers-automatic","slug":"chatgpt-asks-blip-2-answers-automatic","title":"ChatGPT Asks, BLIP-2 Answers: Automatic Questioning Towards Enriched Visual Descriptions","date":"2023-03-12","arxiv_id":"2303.06594","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/chatgpt-asks-blip-2-answers-automatic#ran","syntology_url":"https://syntology.ai/paper/2303.06594","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.06594"}},"official":{"repos":["vision-cair/chatcaptioner"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/spawrious-a-benchmark-for-fine-control-of","slug":"spawrious-a-benchmark-for-fine-control-of","title":"Spawrious: A Benchmark for Fine Control of Spurious Correlation Biases","date":"2023-03-09","arxiv_id":"2303.05470","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/spawrious-a-benchmark-for-fine-control-of#ran","syntology_url":"https://syntology.ai/paper/2303.05470","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.05470"}},"official":{"repos":["aengusl/spawrious"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/conzic-controllable-zero-shot-image","slug":"conzic-controllable-zero-shot-image","title":"ConZIC: Controllable Zero-shot Image Captioning by Sampling-Based Polishing","date":"2023-03-04","arxiv_id":"2303.02437","repositories_listed":1,"syntology":{"n":10,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":7,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/conzic-controllable-zero-shot-image#ran","syntology_url":"https://syntology.ai/paper/2303.02437","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.02437"}},"official":{"repos":["joeyz0z/conzic"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/blip-2-bootstrapping-language-image-pre","slug":"blip-2-bootstrapping-language-image-pre","title":"BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models","date":"2023-01-30","arxiv_id":"2301.12597","repositories_listed":17,"syntology":{"n":8,"n_ran":4,"n_constructed":4,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","sample_list":"/paper/blip-2-bootstrapping-language-image-pre#ran","syntology_url":"https://syntology.ai/paper/2301.12597","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.12597"}},"official":{"repos":["salesforce/lavis"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/noise-aware-learning-from-web-crawled-image","slug":"noise-aware-learning-from-web-crawled-image","title":"Noise-aware Learning from Web-crawled Image-Text Data for Image Captioning","date":"2022-12-27","arxiv_id":"2212.13563","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/noise-aware-learning-from-web-crawled-image#ran","syntology_url":"https://syntology.ai/paper/2212.13563","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.13563"}},"official":{"repos":["kakaobrain/noc"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/position-guided-text-prompt-for-vision","slug":"position-guided-text-prompt-for-vision","title":"Position-guided Text Prompt for Vision-Language Pre-training","date":"2022-12-19","arxiv_id":"2212.09737","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/position-guided-text-prompt-for-vision#ran","syntology_url":"https://syntology.ai/paper/2212.09737","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.09737"}},"official":{"repos":["sail-sg/ptp"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/are-multimodal-models-robust-to-image-and","slug":"are-multimodal-models-robust-to-image-and","title":"Benchmarking Robustness of Multimodal Image-Text Models under Distribution Shift","date":"2022-12-15","arxiv_id":"2212.08044","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/are-multimodal-models-robust-to-image-and#ran","syntology_url":"https://syntology.ai/paper/2212.08044","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.08044"}},"official":null}},{"url":"/paper/adaptive-testing-of-computer-vision-models","slug":"adaptive-testing-of-computer-vision-models","title":"Adaptive Testing of Computer Vision Models","date":"2022-12-06","arxiv_id":"2212.02774","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/adaptive-testing-of-computer-vision-models#ran","syntology_url":"https://syntology.ai/paper/2212.02774","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.02774"}},"official":{"repos":["i-gao/adavision"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/x-2-vlm-all-in-one-pre-trained-model-for","slug":"x-2-vlm-all-in-one-pre-trained-model-for","title":"X$^2$-VLM: All-In-One Pre-trained Model For Vision-Language Tasks","date":"2022-11-22","arxiv_id":"2211.12402","repositories_listed":2,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":6,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/x-2-vlm-all-in-one-pre-trained-model-for#ran","syntology_url":"https://syntology.ai/paper/2211.12402","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.12402"}},"official":{"repos":["zengyan-97/x2-vlm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/exploring-discrete-diffusion-models-for-image","slug":"exploring-discrete-diffusion-models-for-image","title":"Exploring Discrete Diffusion Models for Image Captioning","date":"2022-11-21","arxiv_id":"2211.11694","repositories_listed":1,"syntology":{"n":18,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":4,"n_honours":2,"n_violates":0,"n_no_contract":12,"n_pointer_only":5,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 2 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/exploring-discrete-diffusion-models-for-image#ran","syntology_url":"https://syntology.ai/paper/2211.11694","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.11694"}},"official":{"repos":["buxiangzhiren/ddcap"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/i-can-t-believe-there-s-no-images-learning","slug":"i-can-t-believe-there-s-no-images-learning","title":"I Can't Believe There's No Images! Learning Visual Tasks Using only Language Supervision","date":"2022-11-17","arxiv_id":"2211.09778","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/i-can-t-believe-there-s-no-images-learning#ran","syntology_url":"https://syntology.ai/paper/2211.09778","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.09778"}},"official":{"repos":["allenai/close"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/text-only-training-for-image-captioning-using","slug":"text-only-training-for-image-captioning-using","title":"Text-Only Training for Image Captioning using Noise-Injected CLIP","date":"2022-11-01","arxiv_id":"2211.00575","repositories_listed":4,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/text-only-training-for-image-captioning-using#ran","syntology_url":"https://syntology.ai/paper/2211.00575","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.00575"}},"official":{"repos":["davidhuji/capdec"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/plug-and-play-vqa-zero-shot-vqa-by-conjoining","slug":"plug-and-play-vqa-zero-shot-vqa-by-conjoining","title":"Plug-and-Play VQA: Zero-shot VQA by Conjoining Large Pretrained Models with Zero Training","date":"2022-10-17","arxiv_id":"2210.08773","repositories_listed":3,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/plug-and-play-vqa-zero-shot-vqa-by-conjoining#ran","syntology_url":"https://syntology.ai/paper/2210.08773","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.08773"}},"official":{"repos":["salesforce/lavis"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/mapl-parameter-efficient-adaptation-of","slug":"mapl-parameter-efficient-adaptation-of","title":"MAPL: Parameter-Efficient Adaptation of Unimodal Pre-Trained Models for Vision-Language Few-Shot Prompting","date":"2022-10-13","arxiv_id":"2210.07179","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mapl-parameter-efficient-adaptation-of#ran","syntology_url":"https://syntology.ai/paper/2210.07179","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.07179"}},"official":{"repos":["mair-lab/mapl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-language-maps-for-robot-navigation","slug":"visual-language-maps-for-robot-navigation","title":"Visual Language Maps for Robot Navigation","date":"2022-10-11","arxiv_id":"2210.05714","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/visual-language-maps-for-robot-navigation#ran","syntology_url":"https://syntology.ai/paper/2210.05714","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.05714"}},"official":null}},{"url":"/paper/open-vocabulary-semantic-segmentation-with","slug":"open-vocabulary-semantic-segmentation-with","title":"Open-Vocabulary Semantic Segmentation with Mask-adapted CLIP","date":"2022-10-09","arxiv_id":"2210.04150","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/open-vocabulary-semantic-segmentation-with#ran","syntology_url":"https://syntology.ai/paper/2210.04150","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.04150"}},"official":{"repos":["facebookresearch/ov-seg"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/pix2struct-screenshot-parsing-as-pretraining","slug":"pix2struct-screenshot-parsing-as-pretraining","title":"Pix2Struct: Screenshot Parsing as Pretraining for Visual Language Understanding","date":"2022-10-07","arxiv_id":"2210.03347","repositories_listed":4,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pix2struct-screenshot-parsing-as-pretraining#ran","syntology_url":"https://syntology.ai/paper/2210.03347","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.03347"}},"official":{"repos":["google-research/pix2struct"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-multi-modal-sarcasm-detection-via","slug":"towards-multi-modal-sarcasm-detection-via","title":"Towards Multi-Modal Sarcasm Detection via Hierarchical Congruity Modeling with Knowledge Enhancement","date":"2022-10-07","arxiv_id":"2210.03501","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/towards-multi-modal-sarcasm-detection-via#ran","syntology_url":"https://syntology.ai/paper/2210.03501","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.03501"}},"official":{"repos":["less-and-less-bugs/hkemodel"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/linearly-mapping-from-image-to-text-space","slug":"linearly-mapping-from-image-to-text-space","title":"Linearly Mapping from Image to Text Space","date":"2022-09-30","arxiv_id":"2209.15162","repositories_listed":2,"syntology":{"n":14,"n_ran":13,"n_constructed":0,"n_ran_checked":10,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/linearly-mapping-from-image-to-text-space#ran","syntology_url":"https://syntology.ai/paper/2209.15162","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.15162"}},"official":{"repos":["jmerullo/limber"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/pali-a-jointly-scaled-multilingual-language","slug":"pali-a-jointly-scaled-multilingual-language","title":"PaLI: A Jointly-Scaled Multilingual Language-Image Model","date":"2022-09-14","arxiv_id":"2209.06794","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/pali-a-jointly-scaled-multilingual-language#ran","syntology_url":"https://syntology.ai/paper/2209.06794","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.06794"}},"official":{"repos":["google-research/big_vision"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/gsrformer-grounded-situation-recognition","slug":"gsrformer-grounded-situation-recognition","title":"GSRFormer: Grounded Situation Recognition Transformer with Alternate Semantic Attention Refinement","date":"2022-08-18","arxiv_id":"2208.08965","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":4,"n_honours":2,"n_violates":0,"n_no_contract":3,"n_pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/gsrformer-grounded-situation-recognition#ran","syntology_url":"https://syntology.ai/paper/2208.08965","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.08965"}},"official":{"repos":["zhiqic/gsrformer"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/vault-augmenting-the-vision-and-language","slug":"vault-augmenting-the-vision-and-language","title":"VAuLT: Augmenting the Vision-and-Language Transformer for Sentiment Classification on Social Media","date":"2022-08-18","arxiv_id":"2208.09021","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vault-augmenting-the-vision-and-language#ran","syntology_url":"https://syntology.ai/paper/2208.09021","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.09021"}},"official":{"repos":["gchochla/vault"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/analog-bits-generating-discrete-data-using","slug":"analog-bits-generating-discrete-data-using","title":"Analog Bits: Generating Discrete Data using Diffusion Models with Self-Conditioning","date":"2022-08-08","arxiv_id":"2208.04202","repositories_listed":7,"syntology":{"n":20,"n_ran":13,"n_constructed":0,"n_ran_checked":4,"n_instrument":9,"n_unverified":7,"n_honours":1,"n_violates":3,"n_no_contract":0,"n_pointer_only":4,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 3 violated, 0 with no contract checked; 9 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/analog-bits-generating-discrete-data-using#ran","syntology_url":"https://syntology.ai/paper/2208.04202","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.04202"}},"official":{"repos":["google-research/pix2seq"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/prompt-tuning-for-generative-multimodal","slug":"prompt-tuning-for-generative-multimodal","title":"Prompt Tuning for Generative Multimodal Pretrained Models","date":"2022-08-04","arxiv_id":"2208.02532","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/prompt-tuning-for-generative-multimodal#ran","syntology_url":"https://syntology.ai/paper/2208.02532","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.02532"}},"official":{"repos":["ofa-sys/ofa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/detecting-and-recovering-sequential-deepfake","slug":"detecting-and-recovering-sequential-deepfake","title":"Detecting and Recovering Sequential DeepFake Manipulation","date":"2022-07-05","arxiv_id":"2207.02204","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":2,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/detecting-and-recovering-sequential-deepfake#ran","syntology_url":"https://syntology.ai/paper/2207.02204","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.02204"}},"official":{"repos":["rshaojimmy/seqdeepfake"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/coarse-to-fine-vision-language-pre-training","slug":"coarse-to-fine-vision-language-pre-training","title":"Coarse-to-Fine Vision-Language Pre-training with Fusion in the Backbone","date":"2022-06-15","arxiv_id":"2206.07643","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/coarse-to-fine-vision-language-pre-training#ran","syntology_url":"https://syntology.ai/paper/2206.07643","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.07643"}},"official":{"repos":["microsoft/fiber"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/glipv2-unifying-localization-and-vision","slug":"glipv2-unifying-localization-and-vision","title":"GLIPv2: Unifying Localization and Vision-Language Understanding","date":"2022-06-12","arxiv_id":"2206.05836","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/glipv2-unifying-localization-and-vision#ran","syntology_url":"https://syntology.ai/paper/2206.05836","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.05836"}},"official":{"repos":["microsoft/GLIP"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/git-a-generative-image-to-text-transformer","slug":"git-a-generative-image-to-text-transformer","title":"GIT: A Generative Image-to-text Transformer for Vision and Language","date":"2022-05-27","arxiv_id":"2205.14100","repositories_listed":1,"syntology":{"n":21,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/git-a-generative-image-to-text-transformer#ran","syntology_url":"https://syntology.ai/paper/2205.14100","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.14100"}},"official":{"repos":["microsoft/GenerativeImage2Text"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/mutual-information-divergence-a-unified","slug":"mutual-information-divergence-a-unified","title":"Mutual Information Divergence: A Unified Metric for Multimodal Generative Models","date":"2022-05-25","arxiv_id":"2205.13445","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mutual-information-divergence-a-unified#ran","syntology_url":"https://syntology.ai/paper/2205.13445","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.13445"}},"official":{"repos":["naver-ai/mid.metric"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/language-models-can-see-plugging-visual","slug":"language-models-can-see-plugging-visual","title":"Language Models Can See: Plugging Visual Controls in Text Generation","date":"2022-05-05","arxiv_id":"2205.02655","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/language-models-can-see-plugging-visual#ran","syntology_url":"https://syntology.ai/paper/2205.02655","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.02655"}},"official":{"repos":["yxuansu/magic"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/coca-contrastive-captioners-are-image-text","slug":"coca-contrastive-captioners-are-image-text","title":"CoCa: Contrastive Captioners are Image-Text Foundation Models","date":"2022-05-04","arxiv_id":"2205.01917","repositories_listed":6,"syntology":{"n":17,"n_ran":10,"n_constructed":5,"n_ran_checked":10,"n_instrument":0,"n_unverified":7,"n_honours":2,"n_violates":0,"n_no_contract":8,"n_pointer_only":2,"phrase":"10 ran (of which 5 constructed an object rather than computing a result; 10 with no instrument failure: 2 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/coca-contrastive-captioners-are-image-text#ran","syntology_url":"https://syntology.ai/paper/2205.01917","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.01917"}},"official":null}},{"url":"/paper/end-to-end-transformer-based-model-for-image","slug":"end-to-end-transformer-based-model-for-image","title":"End-to-End Transformer Based Model for Image Captioning","date":"2022-03-29","arxiv_id":"2203.15350","repositories_listed":2,"syntology":{"n":33,"n_ran":17,"n_constructed":9,"n_ran_checked":14,"n_instrument":3,"n_unverified":16,"n_honours":1,"n_violates":2,"n_no_contract":11,"n_pointer_only":15,"phrase":"17 ran (of which 9 constructed an object rather than computing a result; 14 with no instrument failure: 1 honoured, 2 violated, 11 with no contract checked; 3 where Syntology's instrument failed) · 16 unverified","sample_list":"/paper/end-to-end-transformer-based-model-for-image#ran","syntology_url":"https://syntology.ai/paper/2203.15350","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.15350"}},"official":{"repos":["232525/PureT"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":2,"n_ran_no_instrument_failure":5,"n_unverified":7,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/quantifying-societal-bias-amplification-in","slug":"quantifying-societal-bias-amplification-in","title":"Quantifying Societal Bias Amplification in Image Captioning","date":"2022-03-29","arxiv_id":"2203.15395","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/quantifying-societal-bias-amplification-in#ran","syntology_url":"https://syntology.ai/paper/2203.15395","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.15395"}},"official":{"repos":["rebnej/lick-caption-bias"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/linking-emergent-and-natural-languages-via-1","slug":"linking-emergent-and-natural-languages-via-1","title":"Linking Emergent and Natural Languages via Corpus Transfer","date":"2022-03-24","arxiv_id":"2203.13344","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/linking-emergent-and-natural-languages-via-1#ran","syntology_url":"https://syntology.ai/paper/2203.13344","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.13344"}},"official":{"repos":["ysymyth/ec-nl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-vision-features-in-multimodal-machine-1","slug":"on-vision-features-in-multimodal-machine-1","title":"On Vision Features in Multimodal Machine Translation","date":"2022-03-17","arxiv_id":"2203.09173","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/on-vision-features-in-multimodal-machine-1#ran","syntology_url":"https://syntology.ai/paper/2203.09173","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.09173"}},"official":{"repos":["libeineu/fairseq_mmt"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/chart-to-text-a-large-scale-benchmark-for","slug":"chart-to-text-a-large-scale-benchmark-for","title":"Chart-to-Text: A Large-Scale Benchmark for Chart Summarization","date":"2022-03-12","arxiv_id":"2203.06486","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/chart-to-text-a-large-scale-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2203.06486","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.06486"}},"official":null}},{"url":"/paper/geodesic-multi-modal-mixup-for-robust-fine","slug":"geodesic-multi-modal-mixup-for-robust-fine","title":"Geodesic Multi-Modal Mixup for Robust Fine-Tuning","date":"2022-03-08","arxiv_id":"2203.03897","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/geodesic-multi-modal-mixup-for-robust-fine#ran","syntology_url":"https://syntology.ai/paper/2203.03897","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.03897"}},"official":{"repos":["changdaeoh/multimodal-mixup"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/fs-coco-towards-understanding-of-freehand","slug":"fs-coco-towards-understanding-of-freehand","title":"FS-COCO: Towards Understanding of Freehand Sketches of Common Objects in Context","date":"2022-03-04","arxiv_id":"2203.02113","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":5,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/fs-coco-towards-understanding-of-freehand#ran","syntology_url":"https://syntology.ai/paper/2203.02113","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.02113"}},"official":{"repos":["pinakinathc/fscoco"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/dall-eval-probing-the-reasoning-skills-and","slug":"dall-eval-probing-the-reasoning-skills-and","title":"DALL-Eval: Probing the Reasoning Skills and Social Biases of Text-to-Image Generation Models","date":"2022-02-08","arxiv_id":"2202.04053","repositories_listed":2,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":5,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":3,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dall-eval-probing-the-reasoning-skills-and#ran","syntology_url":"https://syntology.ai/paper/2202.04053","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.04053"}},"official":{"repos":["j-min/dalleval"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unifying-architectures-tasks-and-modalities","slug":"unifying-architectures-tasks-and-modalities","title":"OFA: Unifying Architectures, Tasks, and Modalities Through a Simple Sequence-to-Sequence Learning Framework","date":"2022-02-07","arxiv_id":"2202.03052","repositories_listed":4,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unifying-architectures-tasks-and-modalities#ran","syntology_url":"https://syntology.ai/paper/2202.03052","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.03052"}},"official":{"repos":["ofa-sys/ofa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vl-adapter-parameter-efficient-transfer","slug":"vl-adapter-parameter-efficient-transfer","title":"VL-Adapter: Parameter-Efficient Transfer Learning for Vision-and-Language Tasks","date":"2021-12-13","arxiv_id":"2112.06825","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vl-adapter-parameter-efficient-transfer#ran","syntology_url":"https://syntology.ai/paper/2112.06825","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.06825"}},"official":{"repos":["ylsung/vl_adapter"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/injecting-semantic-concepts-into-end-to-end","slug":"injecting-semantic-concepts-into-end-to-end","title":"Injecting Semantic Concepts into End-to-End Image Captioning","date":"2021-12-09","arxiv_id":"2112.05230","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/injecting-semantic-concepts-into-end-to-end#ran","syntology_url":"https://syntology.ai/paper/2112.05230","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.05230"}},"official":{"repos":["jacobswan1/ViTCAP"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/zero-shot-image-to-text-generation-for-visual","slug":"zero-shot-image-to-text-generation-for-visual","title":"ZeroCap: Zero-Shot Image-to-Text Generation for Visual-Semantic Arithmetic","date":"2021-11-29","arxiv_id":"2111.14447","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/zero-shot-image-to-text-generation-for-visual#ran","syntology_url":"https://syntology.ai/paper/2111.14447","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.14447"}},"official":{"repos":["yoadtew/zero-shot-image-to-text"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/crossing-the-format-boundary-of-text-and","slug":"crossing-the-format-boundary-of-text-and","title":"UniTAB: Unifying Text and Box Outputs for Grounded Vision-Language Modeling","date":"2021-11-23","arxiv_id":"2111.12085","repositories_listed":1,"syntology":{"n":15,"n_ran":14,"n_constructed":0,"n_ran_checked":9,"n_instrument":5,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":5,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/crossing-the-format-boundary-of-text-and#ran","syntology_url":"https://syntology.ai/paper/2111.12085","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.12085"}},"official":{"repos":["microsoft/UniTAB"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/clipcap-clip-prefix-for-image-captioning","slug":"clipcap-clip-prefix-for-image-captioning","title":"ClipCap: CLIP Prefix for Image Captioning","date":"2021-11-18","arxiv_id":"2111.09734","repositories_listed":4,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/clipcap-clip-prefix-for-image-captioning#ran","syntology_url":"https://syntology.ai/paper/2111.09734","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.09734"}},"official":{"repos":["rmokady/clip_prefix_caption"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/multi-grained-vision-language-pre-training","slug":"multi-grained-vision-language-pre-training","title":"Multi-Grained Vision Language Pre-Training: Aligning Texts with Visual Concepts","date":"2021-11-16","arxiv_id":"2111.08276","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-grained-vision-language-pre-training#ran","syntology_url":"https://syntology.ai/paper/2111.08276","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.08276"}},"official":{"repos":["zengyan-97/x-vlm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/discovering-non-monotonic-autoregressive","slug":"discovering-non-monotonic-autoregressive","title":"Discovering Non-monotonic Autoregressive Orderings with Variational Inference","date":"2021-10-27","arxiv_id":"2110.15797","repositories_listed":1,"syntology":{"n":7,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/discovering-non-monotonic-autoregressive#ran","syntology_url":"https://syntology.ai/paper/2110.15797","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.15797"}},"official":{"repos":["xuanlinli17/autoregressive_inference"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/end-to-end-supermask-pruning-learning-to","slug":"end-to-end-supermask-pruning-learning-to","title":"End-to-End Supermask Pruning: Learning to Prune Image Captioning Models","date":"2021-10-07","arxiv_id":"2110.03298","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/end-to-end-supermask-pruning-learning-to#ran","syntology_url":"https://syntology.ai/paper/2110.03298","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.03298"}},"official":{"repos":["jiahuei/sparse-image-captioning"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/let-there-be-a-clock-on-the-beach-reducing","slug":"let-there-be-a-clock-on-the-beach-reducing","title":"Let there be a clock on the beach: Reducing Object Hallucination in Image Captioning","date":"2021-10-04","arxiv_id":"2110.01705","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/let-there-be-a-clock-on-the-beach-reducing#ran","syntology_url":"https://syntology.ai/paper/2110.01705","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.01705"}},"official":{"repos":["furkanbiten/object-bias"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/caption-enriched-samples-for-improving","slug":"caption-enriched-samples-for-improving","title":"Caption Enriched Samples for Improving Hateful Memes Detection","date":"2021-09-22","arxiv_id":"2109.10649","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/caption-enriched-samples-for-improving#ran","syntology_url":"https://syntology.ai/paper/2109.10649","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.10649"}},"official":{"repos":["efrat-safanov/caption-enriched-samples-research"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/an-empirical-study-of-gpt-3-for-few-shot","slug":"an-empirical-study-of-gpt-3-for-few-shot","title":"An Empirical Study of GPT-3 for Few-Shot Knowledge-Based VQA","date":"2021-09-10","arxiv_id":"2109.05014","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-empirical-study-of-gpt-3-for-few-shot#ran","syntology_url":"https://syntology.ai/paper/2109.05014","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.05014"}},"official":{"repos":["microsoft/PICa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/simvlm-simple-visual-language-model","slug":"simvlm-simple-visual-language-model","title":"SimVLM: Simple Visual Language Model Pretraining with Weak Supervision","date":"2021-08-24","arxiv_id":"2108.10904","repositories_listed":2,"syntology":{"n":37,"n_ran":22,"n_constructed":8,"n_ran_checked":16,"n_instrument":6,"n_unverified":15,"n_honours":1,"n_violates":3,"n_no_contract":12,"n_pointer_only":32,"phrase":"22 ran (of which 8 constructed an object rather than computing a result; 16 with no instrument failure: 1 honoured, 3 violated, 12 with no contract checked; 6 where Syntology's instrument failed) · 15 unverified","sample_list":"/paper/simvlm-simple-visual-language-model#ran","syntology_url":"https://syntology.ai/paper/2108.10904","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.10904"}},"official":null}},{"url":"/paper/umic-an-unreferenced-metric-for-image","slug":"umic-an-unreferenced-metric-for-image","title":"UMIC: An Unreferenced Metric for Image Captioning via Contrastive Learning","date":"2021-06-26","arxiv_id":"2106.14019","repositories_listed":1,"syntology":{"n":14,"n_ran":7,"n_constructed":4,"n_ran_checked":7,"n_instrument":0,"n_unverified":7,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"7 ran (of which 4 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/umic-an-unreferenced-metric-for-image#ran","syntology_url":"https://syntology.ai/paper/2106.14019","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.14019"}},"official":{"repos":["hwanheelee1993/UMIC"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":4,"n_ran_no_instrument_failure":7,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/understanding-and-evaluating-racial-biases-in","slug":"understanding-and-evaluating-racial-biases-in","title":"Understanding and Evaluating Racial Biases in Image Captioning","date":"2021-06-16","arxiv_id":"2106.08503","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/understanding-and-evaluating-racial-biases-in#ran","syntology_url":"https://syntology.ai/paper/2106.08503","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.08503"}},"official":{"repos":["princetonvisualai/imagecaptioning-bias"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/counterfactual-maximum-likelihood-estimation","slug":"counterfactual-maximum-likelihood-estimation","title":"Counterfactual Maximum Likelihood Estimation for Training Deep Networks","date":"2021-06-07","arxiv_id":"2106.03831","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/counterfactual-maximum-likelihood-estimation#ran","syntology_url":"https://syntology.ai/paper/2106.03831","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.03831"}},"official":{"repos":["WANGXinyiLinda/CMLE"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/smurf-semantic-and-linguistic-understanding","slug":"smurf-semantic-and-linguistic-understanding","title":"SMURF: SeMantic and linguistic UndeRstanding Fusion for Caption Evaluation via Typicality Analysis","date":"2021-06-02","arxiv_id":"2106.01444","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/smurf-semantic-and-linguistic-understanding#ran","syntology_url":"https://syntology.ai/paper/2106.01444","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.01444"}},"official":{"repos":["JoshuaFeinglass/SMURF"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-modal-understanding-and-generation-for","slug":"multi-modal-understanding-and-generation-for","title":"Multi-modal Understanding and Generation for Medical Images and Text via Vision-Language Pre-Training","date":"2021-05-24","arxiv_id":"2105.11333","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multi-modal-understanding-and-generation-for#ran","syntology_url":"https://syntology.ai/paper/2105.11333","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.11333"}},"official":{"repos":["SuperSupermoon/MedViLL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/passage-retrieval-for-outside-knowledge","slug":"passage-retrieval-for-outside-knowledge","title":"Passage Retrieval for Outside-Knowledge Visual Question Answering","date":"2021-05-09","arxiv_id":"2105.03938","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/passage-retrieval-for-outside-knowledge#ran","syntology_url":"https://syntology.ai/paper/2105.03938","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.03938"}},"official":{"repos":["prdwb/okvqa-release"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/human-like-controllable-image-captioning-with","slug":"human-like-controllable-image-captioning-with","title":"Human-like Controllable Image Captioning with Verb-specific Semantic Roles","date":"2021-03-22","arxiv_id":"2103.12204","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/human-like-controllable-image-captioning-with#ran","syntology_url":"https://syntology.ai/paper/2103.12204","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.12204"}},"official":{"repos":["mad-red/VSR-guided-CIC"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/visualgpt-data-efficient-image-captioning-by","slug":"visualgpt-data-efficient-image-captioning-by","title":"VisualGPT: Data-efficient Adaptation of Pretrained Language Models for Image Captioning","date":"2021-02-20","arxiv_id":"2102.10407","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/visualgpt-data-efficient-image-captioning-by#ran","syntology_url":"https://syntology.ai/paper/2102.10407","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.10407"}},"official":{"repos":["Vision-CAIR/VisualGPT"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/sg2caps-revisiting-scene-graphs-for-image","slug":"sg2caps-revisiting-scene-graphs-for-image","title":"In Defense of Scene Graphs for Image Captioning","date":"2021-02-09","arxiv_id":"2102.04990","repositories_listed":1,"syntology":{"n":10,"n_ran":4,"n_constructed":1,"n_ran_checked":1,"n_instrument":3,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":10,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/sg2caps-revisiting-scene-graphs-for-image#ran","syntology_url":"https://syntology.ai/paper/2102.04990","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.04990"}},"official":{"repos":["kien085/sg2caps"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":6,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/unifying-vision-and-language-tasks-via-text","slug":"unifying-vision-and-language-tasks-via-text","title":"Unifying Vision-and-Language Tasks via Text Generation","date":"2021-02-04","arxiv_id":"2102.02779","repositories_listed":2,"syntology":{"n":12,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":10,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 10 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/unifying-vision-and-language-tasks-via-text#ran","syntology_url":"https://syntology.ai/paper/2102.02779","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.02779"}},"official":{"repos":["j-min/VL-T5"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":10,"ran_from_kinds":["official"]}}},{"url":"/paper/vinvl-making-visual-representations-matter-in","slug":"vinvl-making-visual-representations-matter-in","title":"VinVL: Revisiting Visual Representations in Vision-Language Models","date":"2021-01-02","arxiv_id":"2101.00529","repositories_listed":7,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vinvl-making-visual-representations-matter-in#ran","syntology_url":"https://syntology.ai/paper/2101.00529","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2101.00529"}},"official":{"repos":["pzzhang/VinVL"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/generating-image-descriptions-via-sequential","slug":"generating-image-descriptions-via-sequential","title":"Generating Image Descriptions via Sequential Cross-Modal Alignment Guided by Human Gaze","date":"2020-11-09","arxiv_id":"2011.04592","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/generating-image-descriptions-via-sequential#ran","syntology_url":"https://syntology.ai/paper/2011.04592","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.04592"}},"official":{"repos":["dmg-illc/didec-seq-gen"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/diverse-image-captioning-with-context-object","slug":"diverse-image-captioning-with-context-object","title":"Diverse Image Captioning with Context-Object Split Latent Spaces","date":"2020-11-02","arxiv_id":"2011.00966","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/diverse-image-captioning-with-context-object#ran","syntology_url":"https://syntology.ai/paper/2011.00966","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.00966"}},"official":{"repos":["visinf/cos-cvae"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/vokenization-improving-language-understanding","slug":"vokenization-improving-language-understanding","title":"Vokenization: Improving Language Understanding with Contextualized, Visual-Grounded Supervision","date":"2020-10-14","arxiv_id":"2010.06775","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vokenization-improving-language-understanding#ran","syntology_url":"https://syntology.ai/paper/2010.06775","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.06775"}},"official":{"repos":["airsplay/vokenization"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/x-lxmert-paint-caption-and-answer-questions","slug":"x-lxmert-paint-caption-and-answer-questions","title":"X-LXMERT: Paint, Caption and Answer Questions with Multi-Modal Transformers","date":"2020-09-23","arxiv_id":"2009.11278","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/x-lxmert-paint-caption-and-answer-questions#ran","syntology_url":"https://syntology.ai/paper/2009.11278","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.11278"}},"official":{"repos":["allenai/x-lxmert"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/text-as-neural-operator-image-manipulation-by","slug":"text-as-neural-operator-image-manipulation-by","title":"Text as Neural Operator: Image Manipulation by Text Instruction","date":"2020-08-11","arxiv_id":"2008.04556","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/text-as-neural-operator-image-manipulation-by#ran","syntology_url":"https://syntology.ai/paper/2008.04556","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.04556"}},"official":{"repos":["google/tim-gan"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fashion-captioning-towards-generating","slug":"fashion-captioning-towards-generating","title":"Fashion Captioning: Towards Generating Accurate Descriptions with Semantic Rewards","date":"2020-08-06","arxiv_id":"2008.02693","repositories_listed":2,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":4,"n_instrument":4,"n_unverified":0,"n_honours":1,"n_violates":2,"n_no_contract":1,"n_pointer_only":3,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 2 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fashion-captioning-towards-generating#ran","syntology_url":"https://syntology.ai/paper/2008.02693","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.02693"}},"official":{"repos":["xuewyang/Fashion_Captioning"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/comprehensive-image-captioning-via-scene","slug":"comprehensive-image-captioning-via-scene","title":"Comprehensive Image Captioning via Scene Graph Decomposition","date":"2020-07-23","arxiv_id":"2007.11731","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/comprehensive-image-captioning-via-scene#ran","syntology_url":"https://syntology.ai/paper/2007.11731","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.11731"}},"official":null}},{"url":"/paper/graph-optimal-transport-for-cross-domain","slug":"graph-optimal-transport-for-cross-domain","title":"Graph Optimal Transport for Cross-Domain Alignment","date":"2020-06-26","arxiv_id":"2006.14744","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":2,"n_no_contract":6,"n_pointer_only":3,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 2 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/graph-optimal-transport-for-cross-domain#ran","syntology_url":"https://syntology.ai/paper/2006.14744","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.14744"}},"official":{"repos":["LiqunChen0606/Graph-Optimal-Transport"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-image-captioning-with-better-use-of-1","slug":"improving-image-captioning-with-better-use-of-1","title":"Improving Image Captioning with Better Use of Captions","date":"2020-06-21","arxiv_id":"2006.11807","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/improving-image-captioning-with-better-use-of-1#ran","syntology_url":"https://syntology.ai/paper/2006.11807","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.11807"}},"official":{"repos":["Gitsamshi/WeakVRD-Captioning"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mitigating-gender-bias-in-captioning-systems","slug":"mitigating-gender-bias-in-captioning-systems","title":"Mitigating Gender Bias in Captioning Systems","date":"2020-06-15","arxiv_id":"2006.08315","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mitigating-gender-bias-in-captioning-systems#ran","syntology_url":"https://syntology.ai/paper/2006.08315","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.08315"}},"official":{"repos":["datamllab/Mitigating_Gender_Bias_In_Captioning_System"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/virtex-learning-visual-representations-from","slug":"virtex-learning-visual-representations-from","title":"VirTex: Learning Visual Representations from Textual Annotations","date":"2020-06-11","arxiv_id":"2006.06666","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/virtex-learning-visual-representations-from#ran","syntology_url":"https://syntology.ai/paper/2006.06666","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.06666"}},"official":{"repos":["kdexd/virtex"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/demystifying-self-supervised-learning-an","slug":"demystifying-self-supervised-learning-an","title":"Self-supervised Learning from a Multi-view Perspective","date":"2020-06-10","arxiv_id":"2006.05576","repositories_listed":1,"syntology":{"n":15,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/demystifying-self-supervised-learning-an#ran","syntology_url":"https://syntology.ai/paper/2006.05576","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.05576"}},"official":{"repos":["yaohungt/Demystifying_Self_Supervised_Learning"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/dense-caption-matching-and-frame-selection","slug":"dense-caption-matching-and-frame-selection","title":"Dense-Caption Matching and Frame-Selection Gating for Temporal Localization in VideoQA","date":"2020-05-13","arxiv_id":"2005.06409","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":6,"n_ran_checked":8,"n_instrument":1,"n_unverified":3,"n_honours":2,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"9 ran (of which 6 constructed an object rather than computing a result; 8 with no instrument failure: 2 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/dense-caption-matching-and-frame-selection#ran","syntology_url":"https://syntology.ai/paper/2005.06409","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.06409"}},"official":{"repos":["hyounghk/VideoQADenseCapFrameGate-ACL2020"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":6,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/cobra-contrastive-bi-modal-representation","slug":"cobra-contrastive-bi-modal-representation","title":"COBRA: Contrastive Bi-Modal Representation Algorithm","date":"2020-05-07","arxiv_id":"2005.03687","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/cobra-contrastive-bi-modal-representation#ran","syntology_url":"https://syntology.ai/paper/2005.03687","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.03687"}},"official":{"repos":["ovshake/cobra"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/pragmatic-issue-sensitive-image-captioning","slug":"pragmatic-issue-sensitive-image-captioning","title":"Pragmatic Issue-Sensitive Image Captioning","date":"2020-04-29","arxiv_id":"2004.14451","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pragmatic-issue-sensitive-image-captioning#ran","syntology_url":"https://syntology.ai/paper/2004.14451","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.14451"}},"official":{"repos":["windweller/Pragmatic-ISIC"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/transform-and-tell-entity-aware-news-image","slug":"transform-and-tell-entity-aware-news-image","title":"Transform and Tell: Entity-Aware News Image Captioning","date":"2020-04-17","arxiv_id":"2004.08070","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/transform-and-tell-entity-aware-news-image#ran","syntology_url":"https://syntology.ai/paper/2004.08070","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.08070"}},"official":{"repos":["alasdairtran/transform-and-tell"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/oscar-object-semantics-aligned-pre-training","slug":"oscar-object-semantics-aligned-pre-training","title":"Oscar: Object-Semantics Aligned Pre-training for Vision-Language Tasks","date":"2020-04-13","arxiv_id":"2004.06165","repositories_listed":4,"syntology":{"n":23,"n_ran":13,"n_constructed":6,"n_ran_checked":10,"n_instrument":3,"n_unverified":10,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":1,"phrase":"13 ran (of which 6 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 3 where Syntology's instrument failed) · 10 unverified","sample_list":"/paper/oscar-object-semantics-aligned-pre-training#ran","syntology_url":"https://syntology.ai/paper/2004.06165","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.06165"}},"official":{"repos":["microsoft/Oscar"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/more-grounded-image-captioning-by-distilling","slug":"more-grounded-image-captioning-by-distilling","title":"More Grounded Image Captioning by Distilling Image-Text Matching Model","date":"2020-04-01","arxiv_id":"2004.00390","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/more-grounded-image-captioning-by-distilling#ran","syntology_url":"https://syntology.ai/paper/2004.00390","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.00390"}},"official":{"repos":["YuanEZhou/Grounded-Image-Captioning"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/egoshots-an-ego-vision-life-logging-dataset","slug":"egoshots-an-ego-vision-life-logging-dataset","title":"Egoshots, an ego-vision life-logging dataset and semantic fidelity metric to evaluate diversity in image captioning models","date":"2020-03-26","arxiv_id":"2003.11743","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/egoshots-an-ego-vision-life-logging-dataset#ran","syntology_url":"https://syntology.ai/paper/2003.11743","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.11743"}},"official":{"repos":["NataliaDiaz/Egoshots","Pranav21091996/Semantic_Fidelity-and-Egoshots"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/say-as-you-wish-fine-grained-control-of-image","slug":"say-as-you-wish-fine-grained-control-of-image","title":"Say As You Wish: Fine-grained Control of Image Caption Generation with Abstract Scene Graphs","date":"2020-03-01","arxiv_id":"2003.00387","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/say-as-you-wish-fine-grained-control-of-image#ran","syntology_url":"https://syntology.ai/paper/2003.00387","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.00387"}},"official":null}},{"url":"/paper/visual-commonsense-r-cnn","slug":"visual-commonsense-r-cnn","title":"Visual Commonsense R-CNN","date":"2020-02-27","arxiv_id":"2002.12204","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/visual-commonsense-r-cnn#ran","syntology_url":"https://syntology.ai/paper/2002.12204","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.12204"}},"official":{"repos":["Wangt-CN/VC-R-CNN"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/latent-normalizing-flows-for-many-to-many-1","slug":"latent-normalizing-flows-for-many-to-many-1","title":"Latent Normalizing Flows for Many-to-Many Cross-Domain Mappings","date":"2020-02-16","arxiv_id":"2002.06661","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/latent-normalizing-flows-for-many-to-many-1#ran","syntology_url":"https://syntology.ai/paper/2002.06661","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.06661"}},"official":{"repos":["visinf/lnfmm"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/adapting-grad-cam-for-embedding-networks","slug":"adapting-grad-cam-for-embedding-networks","title":"Adapting Grad-CAM for Embedding Networks","date":"2020-01-17","arxiv_id":"2001.06538","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/adapting-grad-cam-for-embedding-networks#ran","syntology_url":"https://syntology.ai/paper/2001.06538","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.06538"}},"official":null}},{"url":"/paper/explicit-sparse-transformer-concentrated","slug":"explicit-sparse-transformer-concentrated","title":"Explicit Sparse Transformer: Concentrated Attention Through Explicit Selection","date":"2019-12-25","arxiv_id":"1912.11637","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":1,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/explicit-sparse-transformer-concentrated#ran","syntology_url":"https://syntology.ai/paper/1912.11637","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.11637"}},"official":{"repos":["lancopku/Explicit-Sparse-Transformer"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/m2-meshed-memory-transformer-for-image","slug":"m2-meshed-memory-transformer-for-image","title":"Meshed-Memory Transformer for Image Captioning","date":"2019-12-17","arxiv_id":"1912.08226","repositories_listed":2,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/m2-meshed-memory-transformer-for-image#ran","syntology_url":"https://syntology.ai/paper/1912.08226","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.08226"}},"official":{"repos":["aimagelab/meshed-memory-transformer"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-fairness-in-visual-recognition","slug":"towards-fairness-in-visual-recognition","title":"Towards Fairness in Visual Recognition: Effective Strategies for Bias Mitigation","date":"2019-11-26","arxiv_id":"1911.11834","repositories_listed":3,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/towards-fairness-in-visual-recognition#ran","syntology_url":"https://syntology.ai/paper/1911.11834","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.11834"}},"official":{"repos":["princetonvisualai/DomainBiasMitigation"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/sequence-modeling-with-unconstrained","slug":"sequence-modeling-with-unconstrained","title":"Sequence Modeling with Unconstrained Generation Order","date":"2019-11-01","arxiv_id":"1911.00176","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":9,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 1 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/sequence-modeling-with-unconstrained#ran","syntology_url":"https://syntology.ai/paper/1911.00176","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.00176"}},"official":{"repos":["TIXFeniks/neurips2019_intrus"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}}],"record_sha256":"7de56b3eca46b5d3d71ac5ae6425bd9ef8bcacf5e9a34dae5a657a8a5292cbb0","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}