{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/image-captioning/papers/6","list_of":"/task/image-captioning","task":"Image Captioning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":6,"pages_in_order":19,"rows_per_page":100,"rows":[501,600],"of":1878,"counts":{"archive_papers_tagged":1878,"with_a_code_link":774,"where_syntology_ran_a_sample":243,"not_listed_spam_title":0,"listed":1878,"listed_where_code_ran":243,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":201,"every_run_a_failure_of_syntologys_instrument":42,"listed_with_a_run_with_no_instrument_failure":201,"listed_every_run_a_failure_of_syntologys_instrument":42,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/image-captioning","prev":"/task/image-captioning/papers/5","next":"/task/image-captioning/papers/7","papers":[{"url":"/paper/glipv2-unifying-localization-and-vision","slug":"glipv2-unifying-localization-and-vision","title":"GLIPv2: Unifying Localization and Vision-Language Understanding","date":"2022-06-12","arxiv_id":"2206.05836","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/glipv2-unifying-localization-and-vision#ran","syntology_url":"https://syntology.ai/paper/2206.05836","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.05836"}},"official":{"repos":["microsoft/GLIP"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/uni-perceiver-moe-learning-sparse-generalist","slug":"uni-perceiver-moe-learning-sparse-generalist","title":"Uni-Perceiver-MoE: Learning Sparse Generalist Models with Conditional MoEs","date":"2022-06-09","arxiv_id":"2206.04674","repositories_listed":1,"syntology":null},{"url":"/paper/expressive-scene-graph-generation-using","slug":"expressive-scene-graph-generation-using","title":"Expressive Scene Graph Generation Using Commonsense Knowledge Infusion for Visual Understanding and Reasoning","date":"2022-05-31","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/ban-cap-a-multi-purpose-english-bangla-image","slug":"ban-cap-a-multi-purpose-english-bangla-image","title":"BAN-Cap: A Multi-Purpose English-Bangla Image Descriptions Dataset","date":"2022-05-28","arxiv_id":"2205.14462","repositories_listed":1,"syntology":null},{"url":"/paper/variational-transformer-a-framework-beyond","slug":"variational-transformer-a-framework-beyond","title":"Variational Transformer: A Framework Beyond the Trade-off between Accuracy and Diversity for Image Captioning","date":"2022-05-28","arxiv_id":"2205.14458","repositories_listed":1,"syntology":null},{"url":"/paper/git-a-generative-image-to-text-transformer","slug":"git-a-generative-image-to-text-transformer","title":"GIT: A Generative Image-to-text Transformer for Vision and Language","date":"2022-05-27","arxiv_id":"2205.14100","repositories_listed":1,"syntology":{"n":21,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/git-a-generative-image-to-text-transformer#ran","syntology_url":"https://syntology.ai/paper/2205.14100","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.14100"}},"official":{"repos":["microsoft/GenerativeImage2Text"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/fine-grained-image-captioning-with-clip","slug":"fine-grained-image-captioning-with-clip","title":"Fine-grained Image Captioning with CLIP Reward","date":"2022-05-26","arxiv_id":"2205.13115","repositories_listed":1,"syntology":null},{"url":"/paper/mutual-information-divergence-a-unified","slug":"mutual-information-divergence-a-unified","title":"Mutual Information Divergence: A Unified Metric for Multimodal Generative Models","date":"2022-05-25","arxiv_id":"2205.13445","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mutual-information-divergence-a-unified#ran","syntology_url":"https://syntology.ai/paper/2205.13445","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.13445"}},"official":{"repos":["naver-ai/mid.metric"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-greedy-search-tracking-by-multi-agent","slug":"beyond-greedy-search-tracking-by-multi-agent","title":"Beyond Greedy Search: Tracking by Multi-Agent Reinforcement Learning-based Beam Search","date":"2022-05-19","arxiv_id":"2205.09676","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-a-pre-trained-object-detector-cross","slug":"beyond-a-pre-trained-object-detector-cross","title":"Beyond a Pre-Trained Object Detector: Cross-Modal Textual and Visual Context for Image Captioning","date":"2022-05-09","arxiv_id":"2205.04363","repositories_listed":1,"syntology":null},{"url":"/paper/language-models-can-see-plugging-visual","slug":"language-models-can-see-plugging-visual","title":"Language Models Can See: Plugging Visual Controls in Text Generation","date":"2022-05-05","arxiv_id":"2205.02655","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/language-models-can-see-plugging-visual#ran","syntology_url":"https://syntology.ai/paper/2205.02655","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.02655"}},"official":{"repos":["yxuansu/magic"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/diverse-image-captioning-with-grounded-style","slug":"diverse-image-captioning-with-grounded-style","title":"Diverse Image Captioning with Grounded Style","date":"2022-05-03","arxiv_id":"2205.01813","repositories_listed":1,"syntology":null},{"url":"/paper/image-captioning-in-the-transformer-age","slug":"image-captioning-in-the-transformer-age","title":"Image Captioning In the Transformer Age","date":"2022-04-15","arxiv_id":"2204.07374","repositories_listed":1,"syntology":null},{"url":"/paper/socratic-models-composing-zero-shot","slug":"socratic-models-composing-zero-shot","title":"Socratic Models: Composing Zero-Shot Multimodal Reasoning with Language","date":"2022-04-01","arxiv_id":"2204.00598","repositories_listed":1,"syntology":null},{"url":"/paper/quantifying-societal-bias-amplification-in","slug":"quantifying-societal-bias-amplification-in","title":"Quantifying Societal Bias Amplification in Image Captioning","date":"2022-03-29","arxiv_id":"2203.15395","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/quantifying-societal-bias-amplification-in#ran","syntology_url":"https://syntology.ai/paper/2203.15395","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.15395"}},"official":{"repos":["rebnej/lick-caption-bias"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/linking-emergent-and-natural-languages-via-1","slug":"linking-emergent-and-natural-languages-via-1","title":"Linking Emergent and Natural Languages via Corpus Transfer","date":"2022-03-24","arxiv_id":"2203.13344","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/linking-emergent-and-natural-languages-via-1#ran","syntology_url":"https://syntology.ai/paper/2203.13344","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.13344"}},"official":{"repos":["ysymyth/ec-nl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/geodesic-multi-modal-mixup-for-robust-fine","slug":"geodesic-multi-modal-mixup-for-robust-fine","title":"Geodesic Multi-Modal Mixup for Robust Fine-Tuning","date":"2022-03-08","arxiv_id":"2203.03897","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/geodesic-multi-modal-mixup-for-robust-fine#ran","syntology_url":"https://syntology.ai/paper/2203.03897","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.03897"}},"official":{"repos":["changdaeoh/multimodal-mixup"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/fs-coco-towards-understanding-of-freehand","slug":"fs-coco-towards-understanding-of-freehand","title":"FS-COCO: Towards Understanding of Freehand Sketches of Common Objects in Context","date":"2022-03-04","arxiv_id":"2203.02113","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":5,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/fs-coco-towards-understanding-of-freehand#ran","syntology_url":"https://syntology.ai/paper/2203.02113","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.02113"}},"official":{"repos":["pinakinathc/fscoco"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/camel-mean-teacher-learning-for-image","slug":"camel-mean-teacher-learning-for-image","title":"CaMEL: Mean Teacher Learning for Image Captioning","date":"2022-02-21","arxiv_id":"2202.10492","repositories_listed":1,"syntology":null},{"url":"/paper/acort-a-compact-object-relation-transformer","slug":"acort-a-compact-object-relation-transformer","title":"ACORT: A Compact Object Relation Transformer for Parameter Efficient Image Captioning","date":"2022-02-11","arxiv_id":"2202.05451","repositories_listed":1,"syntology":null},{"url":"/paper/visual-information-guided-zero-shot","slug":"visual-information-guided-zero-shot","title":"Visual Information Guided Zero-Shot Paraphrase Generation","date":"2022-01-22","arxiv_id":"2201.09107","repositories_listed":1,"syntology":null},{"url":"/paper/compact-bidirectional-transformer-for-image","slug":"compact-bidirectional-transformer-for-image","title":"Compact Bidirectional Transformer for Image Captioning","date":"2022-01-06","arxiv_id":"2201.01984","repositories_listed":1,"syntology":null},{"url":"/paper/deecap-dynamic-early-exiting-for-efficient","slug":"deecap-dynamic-early-exiting-for-efficient","title":"DeeCap: Dynamic Early Exiting for Efficient Image Captioning","date":"2022-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/show-deconfound-and-tell-image-captioning","slug":"show-deconfound-and-tell-image-captioning","title":"Show, Deconfound and Tell: Image Captioning With Causal Inference","date":"2022-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/vl-adapter-parameter-efficient-transfer","slug":"vl-adapter-parameter-efficient-transfer","title":"VL-Adapter: Parameter-Efficient Transfer Learning for Vision-and-Language Tasks","date":"2021-12-13","arxiv_id":"2112.06825","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vl-adapter-parameter-efficient-transfer#ran","syntology_url":"https://syntology.ai/paper/2112.06825","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.06825"}},"official":{"repos":["ylsung/vl_adapter"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/injecting-semantic-concepts-into-end-to-end","slug":"injecting-semantic-concepts-into-end-to-end","title":"Injecting Semantic Concepts into End-to-End Image Captioning","date":"2021-12-09","arxiv_id":"2112.05230","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/injecting-semantic-concepts-into-end-to-end#ran","syntology_url":"https://syntology.ai/paper/2112.05230","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.05230"}},"official":{"repos":["jacobswan1/ViTCAP"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/protecting-intellectual-property-of-language","slug":"protecting-intellectual-property-of-language","title":"Protecting Intellectual Property of Language Generation APIs with Lexical Watermark","date":"2021-12-05","arxiv_id":"2112.02701","repositories_listed":1,"syntology":null},{"url":"/paper/object-centric-unsupervised-image-captioning","slug":"object-centric-unsupervised-image-captioning","title":"Object-Centric Unsupervised Image Captioning","date":"2021-12-02","arxiv_id":"2112.00969","repositories_listed":1,"syntology":null},{"url":"/paper/image2tweet-datasets-in-hindi-and-english-for","slug":"image2tweet-datasets-in-hindi-and-english-for","title":"Image2tweet: Datasets in Hindi and English for Generating Tweets from Images","date":"2021-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/set-prediction-in-the-latent-space","slug":"set-prediction-in-the-latent-space","title":"Set Prediction in the Latent Space","date":"2021-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/zero-shot-image-to-text-generation-for-visual","slug":"zero-shot-image-to-text-generation-for-visual","title":"ZeroCap: Zero-Shot Image-to-Text Generation for Visual-Semantic Arithmetic","date":"2021-11-29","arxiv_id":"2111.14447","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/zero-shot-image-to-text-generation-for-visual#ran","syntology_url":"https://syntology.ai/paper/2111.14447","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.14447"}},"official":{"repos":["yoadtew/zero-shot-image-to-text"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/crossing-the-format-boundary-of-text-and","slug":"crossing-the-format-boundary-of-text-and","title":"UniTAB: Unifying Text and Box Outputs for Grounded Vision-Language Modeling","date":"2021-11-23","arxiv_id":"2111.12085","repositories_listed":1,"syntology":{"n":15,"n_ran":14,"n_constructed":0,"n_ran_checked":9,"n_instrument":5,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":5,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/crossing-the-format-boundary-of-text-and#ran","syntology_url":"https://syntology.ai/paper/2111.12085","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.12085"}},"official":{"repos":["microsoft/UniTAB"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/l-verse-bidirectional-generation-between","slug":"l-verse-bidirectional-generation-between","title":"L-Verse: Bidirectional Generation Between Image and Text","date":"2021-11-22","arxiv_id":"2111.11133","repositories_listed":1,"syntology":null},{"url":"/paper/multi-grained-vision-language-pre-training","slug":"multi-grained-vision-language-pre-training","title":"Multi-Grained Vision Language Pre-Training: Aligning Texts with Visual Concepts","date":"2021-11-16","arxiv_id":"2111.08276","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-grained-vision-language-pre-training#ran","syntology_url":"https://syntology.ai/paper/2111.08276","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.08276"}},"official":{"repos":["zengyan-97/x-vlm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/machine-in-the-loop-rewriting-for-creative","slug":"machine-in-the-loop-rewriting-for-creative","title":"Machine-in-the-Loop Rewriting for Creative Image Captioning","date":"2021-11-07","arxiv_id":"2111.04193","repositories_listed":1,"syntology":null},{"url":"/paper/discovering-non-monotonic-autoregressive","slug":"discovering-non-monotonic-autoregressive","title":"Discovering Non-monotonic Autoregressive Orderings with Variational Inference","date":"2021-10-27","arxiv_id":"2110.15797","repositories_listed":1,"syntology":{"n":7,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/discovering-non-monotonic-autoregressive#ran","syntology_url":"https://syntology.ai/paper/2110.15797","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.15797"}},"official":{"repos":["xuanlinli17/autoregressive_inference"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/bangla-image-caption-generation-through-cnn","slug":"bangla-image-caption-generation-through-cnn","title":"Bangla Image Caption Generation through CNN-Transformer based Encoder-Decoder Network","date":"2021-10-24","arxiv_id":"2110.12442","repositories_listed":1,"syntology":null},{"url":"/paper/scicap-generating-captions-for-scientific","slug":"scicap-generating-captions-for-scientific","title":"SciCap: Generating Captions for Scientific Figures","date":"2021-10-22","arxiv_id":"2110.11624","repositories_listed":1,"syntology":null},{"url":"/paper/a-good-prompt-is-worth-millions-of-parameters","slug":"a-good-prompt-is-worth-millions-of-parameters","title":"A Good Prompt Is Worth Millions of Parameters: Low-resource Prompt-based Learning for Vision-Language Models","date":"2021-10-16","arxiv_id":"2110.08484","repositories_listed":1,"syntology":null},{"url":"/paper/semi-autoregressive-image-captioning","slug":"semi-autoregressive-image-captioning","title":"Semi-Autoregressive Image Captioning","date":"2021-10-11","arxiv_id":"2110.05342","repositories_listed":1,"syntology":null},{"url":"/paper/can-audio-captions-be-evaluated-with-image","slug":"can-audio-captions-be-evaluated-with-image","title":"Can Audio Captions Be Evaluated with Image Caption Metrics?","date":"2021-10-10","arxiv_id":"2110.04684","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/can-audio-captions-be-evaluated-with-image#ran","syntology_url":"https://syntology.ai/paper/2110.04684","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.04684"}},"official":{"repos":["blmoistawinde/fense"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/end-to-end-supermask-pruning-learning-to","slug":"end-to-end-supermask-pruning-learning-to","title":"End-to-End Supermask Pruning: Learning to Prune Image Captioning Models","date":"2021-10-07","arxiv_id":"2110.03298","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/end-to-end-supermask-pruning-learning-to#ran","syntology_url":"https://syntology.ai/paper/2110.03298","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.03298"}},"official":{"repos":["jiahuei/sparse-image-captioning"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/let-there-be-a-clock-on-the-beach-reducing","slug":"let-there-be-a-clock-on-the-beach-reducing","title":"Let there be a clock on the beach: Reducing Object Hallucination in Image Captioning","date":"2021-10-04","arxiv_id":"2110.01705","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/let-there-be-a-clock-on-the-beach-reducing#ran","syntology_url":"https://syntology.ai/paper/2110.01705","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.01705"}},"official":{"repos":["furkanbiten/object-bias"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/geometry-attention-transformer-with-position","slug":"geometry-attention-transformer-with-position","title":"Geometry Attention Transformer with Position-aware LSTMs for Image Captioning","date":"2021-10-01","arxiv_id":"2110.00335","repositories_listed":1,"syntology":null},{"url":"/paper/caption-enriched-samples-for-improving","slug":"caption-enriched-samples-for-improving","title":"Caption Enriched Samples for Improving Hateful Memes Detection","date":"2021-09-22","arxiv_id":"2109.10649","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/caption-enriched-samples-for-improving#ran","syntology_url":"https://syntology.ai/paper/2109.10649","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.10649"}},"official":{"repos":["efrat-safanov/caption-enriched-samples-research"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/label-attention-transformer-with","slug":"label-attention-transformer-with","title":"Label-Attention Transformer with Geometrically Coherent Objects for Image Captioning","date":"2021-09-16","arxiv_id":"2109.07799","repositories_listed":1,"syntology":null},{"url":"/paper/image-captioning-for-effective-use-of","slug":"image-captioning-for-effective-use-of","title":"Image Captioning for Effective Use of Language Models in Knowledge-Based Visual Question Answering","date":"2021-09-15","arxiv_id":"2109.08029","repositories_listed":1,"syntology":null},{"url":"/paper/an-empirical-study-of-gpt-3-for-few-shot","slug":"an-empirical-study-of-gpt-3-for-few-shot","title":"An Empirical Study of GPT-3 for Few-Shot Knowledge-Based VQA","date":"2021-09-10","arxiv_id":"2109.05014","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-empirical-study-of-gpt-3-for-few-shot#ran","syntology_url":"https://syntology.ai/paper/2109.05014","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.05014"}},"official":{"repos":["microsoft/PICa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vision-matters-when-it-should-sanity-checking","slug":"vision-matters-when-it-should-sanity-checking","title":"Vision Matters When It Should: Sanity Checking Multimodal Machine Translation Models","date":"2021-09-08","arxiv_id":"2109.03415","repositories_listed":1,"syntology":null},{"url":"/paper/journalistic-guidelines-aware-news-image","slug":"journalistic-guidelines-aware-news-image","title":"Journalistic Guidelines Aware News Image Captioning","date":"2021-09-07","arxiv_id":"2109.02865","repositories_listed":1,"syntology":null},{"url":"/paper/geneannotator-a-semi-automatic-annotation","slug":"geneannotator-a-semi-automatic-annotation","title":"GeneAnnotator: A Semi-automatic Annotation Tool for Visual Scene Graph","date":"2021-09-06","arxiv_id":"2109.02226","repositories_listed":1,"syntology":null},{"url":"/paper/laviter-learning-aligned-visual-and-textual","slug":"laviter-learning-aligned-visual-and-textual","title":"LAViTeR: Learning Aligned Visual and Textual Representations Assisted by Image and Caption Generation","date":"2021-09-04","arxiv_id":"2109.04993","repositories_listed":1,"syntology":null},{"url":"/paper/zero-shot-natural-language-video-localization","slug":"zero-shot-natural-language-video-localization","title":"Zero-shot Natural Language Video Localization","date":"2021-08-29","arxiv_id":"2110.00428","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-human-attention-in-novel-object","slug":"leveraging-human-attention-in-novel-object","title":"Leveraging Human Attention in Novel Object Captioning","date":"2021-08-19","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/perturb-predict-paraphrase-semi-supervised","slug":"perturb-predict-paraphrase-semi-supervised","title":"Perturb, Predict & Paraphrase: Semi-Supervised Learning using Noisy Student for Image Captioning","date":"2021-08-19","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/towers-of-babel-combining-images-language-and","slug":"towers-of-babel-combining-images-language-and","title":"Towers of Babel: Combining Images, Language, and 3D Geometry for Learning Multimodal Vision","date":"2021-08-12","arxiv_id":"2108.05863","repositories_listed":1,"syntology":null},{"url":"/paper/dual-graph-convolutional-networks-with","slug":"dual-graph-convolutional-networks-with","title":"Dual Graph Convolutional Networks with Transformer and Curriculum Learning for Image Captioning","date":"2021-08-05","arxiv_id":"2108.02366","repositories_listed":1,"syntology":null},{"url":"/paper/neural-twins-talk-alternative-calculations","slug":"neural-twins-talk-alternative-calculations","title":"Neural Twins Talk & Alternative Calculations","date":"2021-08-05","arxiv_id":"2108.02807","repositories_listed":1,"syntology":null},{"url":"/paper/icecap-information-concentrated-entity-aware","slug":"icecap-information-concentrated-entity-aware","title":"ICECAP: Information Concentrated Entity-aware Image Captioning","date":"2021-08-04","arxiv_id":"2108.02050","repositories_listed":1,"syntology":null},{"url":"/paper/question-controlled-text-aware-image","slug":"question-controlled-text-aware-image","title":"Question-controlled Text-aware Image Captioning","date":"2021-08-04","arxiv_id":"2108.02059","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-descriptive-image-captioning-with","slug":"enhancing-descriptive-image-captioning-with","title":"Enhancing Descriptive Image Captioning with Natural Language Inference","date":"2021-08-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/reformer-the-relational-transformer-for-image","slug":"reformer-the-relational-transformer-for-image","title":"ReFormer: The Relational Transformer for Image Captioning","date":"2021-07-29","arxiv_id":"2107.14178","repositories_listed":1,"syntology":null},{"url":"/paper/global-object-proposals-for-improving-multi","slug":"global-object-proposals-for-improving-multi","title":"Global Object Proposals for Improving Multi-Sentence Video Descriptions","date":"2021-07-18","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/umic-an-unreferenced-metric-for-image","slug":"umic-an-unreferenced-metric-for-image","title":"UMIC: An Unreferenced Metric for Image Captioning via Contrastive Learning","date":"2021-06-26","arxiv_id":"2106.14019","repositories_listed":1,"syntology":{"n":14,"n_ran":7,"n_constructed":4,"n_ran_checked":7,"n_instrument":0,"n_unverified":7,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"7 ran (of which 4 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/umic-an-unreferenced-metric-for-image#ran","syntology_url":"https://syntology.ai/paper/2106.14019","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.14019"}},"official":{"repos":["hwanheelee1993/UMIC"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":4,"n_ran_no_instrument_failure":7,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/neural-fashion-image-captioning-accounting","slug":"neural-fashion-image-captioning-accounting","title":"Neural Fashion Image Captioning : Accounting for Data Diversity","date":"2021-06-23","arxiv_id":"2106.12154","repositories_listed":1,"syntology":null},{"url":"/paper/rstnet-captioning-with-adaptive-attention-on","slug":"rstnet-captioning-with-adaptive-attention-on","title":"RSTNet: Captioning With Adaptive Attention on Visual and Non-Visual Words","date":"2021-06-19","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/semi-autoregressive-transformer-for-image","slug":"semi-autoregressive-transformer-for-image","title":"Semi-Autoregressive Transformer for Image Captioning","date":"2021-06-17","arxiv_id":"2106.09436","repositories_listed":1,"syntology":null},{"url":"/paper/understanding-and-evaluating-racial-biases-in","slug":"understanding-and-evaluating-racial-biases-in","title":"Understanding and Evaluating Racial Biases in Image Captioning","date":"2021-06-16","arxiv_id":"2106.08503","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/understanding-and-evaluating-racial-biases-in#ran","syntology_url":"https://syntology.ai/paper/2106.08503","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.08503"}},"official":{"repos":["princetonvisualai/imagecaptioning-bias"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bertgen-multi-task-generation-through-bert","slug":"bertgen-multi-task-generation-through-bert","title":"BERTGEN: Multi-task Generation through BERT","date":"2021-06-07","arxiv_id":"2106.03484","repositories_listed":1,"syntology":null},{"url":"/paper/counterfactual-maximum-likelihood-estimation","slug":"counterfactual-maximum-likelihood-estimation","title":"Counterfactual Maximum Likelihood Estimation for Training Deep Networks","date":"2021-06-07","arxiv_id":"2106.03831","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/counterfactual-maximum-likelihood-estimation#ran","syntology_url":"https://syntology.ai/paper/2106.03831","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.03831"}},"official":{"repos":["WANGXinyiLinda/CMLE"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/smurf-semantic-and-linguistic-understanding","slug":"smurf-semantic-and-linguistic-understanding","title":"SMURF: SeMantic and linguistic UndeRstanding Fusion for Caption Evaluation via Typicality Analysis","date":"2021-06-02","arxiv_id":"2106.01444","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/smurf-semantic-and-linguistic-understanding#ran","syntology_url":"https://syntology.ai/paper/2106.01444","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.01444"}},"official":{"repos":["JoshuaFeinglass/SMURF"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-modal-understanding-and-generation-for","slug":"multi-modal-understanding-and-generation-for","title":"Multi-modal Understanding and Generation for Medical Images and Text via Vision-Language Pre-Training","date":"2021-05-24","arxiv_id":"2105.11333","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multi-modal-understanding-and-generation-for#ran","syntology_url":"https://syntology.ai/paper/2105.11333","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.11333"}},"official":{"repos":["SuperSupermoon/MedViLL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/connecting-what-to-say-with-where-to-look-by","slug":"connecting-what-to-say-with-where-to-look-by","title":"Connecting What to Say With Where to Look by Modeling Human Attention Traces","date":"2021-05-12","arxiv_id":"2105.05964","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/connecting-what-to-say-with-where-to-look-by#ran","syntology_url":"https://syntology.ai/paper/2105.05964","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.05964"}},"official":{"repos":["facebookresearch/connect-caption-and-trace"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/a-hybrid-model-for-combining-neural-image","slug":"a-hybrid-model-for-combining-neural-image","title":"A Hybrid Model for Combining Neural Image Caption and k-Nearest Neighbor Approach for Image Captioning","date":"2021-05-09","arxiv_id":"2105.03826","repositories_listed":1,"syntology":null},{"url":"/paper/passage-retrieval-for-outside-knowledge","slug":"passage-retrieval-for-outside-knowledge","title":"Passage Retrieval for Outside-Knowledge Visual Question Answering","date":"2021-05-09","arxiv_id":"2105.03938","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/passage-retrieval-for-outside-knowledge#ran","syntology_url":"https://syntology.ai/paper/2105.03938","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.03938"}},"official":{"repos":["prdwb/okvqa-release"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/removing-word-level-spurious-alignment","slug":"removing-word-level-spurious-alignment","title":"Removing Word-Level Spurious Alignment between Images and Pseudo-Captions in Unsupervised Image Captioning","date":"2021-04-28","arxiv_id":"2104.13872","repositories_listed":1,"syntology":null},{"url":"/paper/reltransformer-balancing-the-visual","slug":"reltransformer-balancing-the-visual","title":"RelTransformer: A Transformer-Based Long-Tail Visual Relationship Recognition","date":"2021-04-24","arxiv_id":"2104.11934","repositories_listed":1,"syntology":null},{"url":"/paper/towards-accurate-text-based-image-captioning","slug":"towards-accurate-text-based-image-captioning","title":"Towards Accurate Text-based Image Captioning with Content Diversity Exploration","date":"2021-04-23","arxiv_id":"2105.03236","repositories_listed":1,"syntology":null},{"url":"/paper/concadia-tackling-image-accessibility-with","slug":"concadia-tackling-image-accessibility-with","title":"Concadia: Towards Image-Based Text Generation with a Purpose","date":"2021-04-16","arxiv_id":"2104.08376","repositories_listed":1,"syntology":null},{"url":"/paper/wikily-neural-machine-translation-tailored-to","slug":"wikily-neural-machine-translation-tailored-to","title":"\"Wikily\" Supervised Neural Translation Tailored to Cross-Lingual Tasks","date":"2021-04-16","arxiv_id":"2104.08384","repositories_listed":1,"syntology":null},{"url":"/paper/human-like-controllable-image-captioning-with","slug":"human-like-controllable-image-captioning-with","title":"Human-like Controllable Image Captioning with Verb-specific Semantic Roles","date":"2021-03-22","arxiv_id":"2103.12204","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/human-like-controllable-image-captioning-with#ran","syntology_url":"https://syntology.ai/paper/2103.12204","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.12204"}},"official":{"repos":["mad-red/VSR-guided-CIC"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multiple-instance-captioning-learning","slug":"multiple-instance-captioning-learning","title":"Multiple Instance Captioning: Learning Representations from Histopathology Textbooks and Articles","date":"2021-03-08","arxiv_id":"2103.05121","repositories_listed":1,"syntology":null},{"url":"/paper/visual-question-answering-which-investigated","slug":"visual-question-answering-which-investigated","title":"Visual Question Answering: which investigated applications?","date":"2021-03-04","arxiv_id":"2103.02937","repositories_listed":1,"syntology":null},{"url":"/paper/retrieval-augmentation-to-improve-robustness","slug":"retrieval-augmentation-to-improve-robustness","title":"Retrieval Augmentation for Deep Neural Networks","date":"2021-02-25","arxiv_id":"2102.13030","repositories_listed":1,"syntology":null},{"url":"/paper/visualgpt-data-efficient-image-captioning-by","slug":"visualgpt-data-efficient-image-captioning-by","title":"VisualGPT: Data-efficient Adaptation of Pretrained Language Models for Image Captioning","date":"2021-02-20","arxiv_id":"2102.10407","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/visualgpt-data-efficient-image-captioning-by#ran","syntology_url":"https://syntology.ai/paper/2102.10407","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.10407"}},"official":{"repos":["Vision-CAIR/VisualGPT"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/improved-bengali-image-captioning-via-deep","slug":"improved-bengali-image-captioning-via-deep","title":"Improved Bengali Image Captioning via deep convolutional neural network based encoder-decoder model","date":"2021-02-14","arxiv_id":"2102.07192","repositories_listed":1,"syntology":null},{"url":"/paper/sg2caps-revisiting-scene-graphs-for-image","slug":"sg2caps-revisiting-scene-graphs-for-image","title":"In Defense of Scene Graphs for Image Captioning","date":"2021-02-09","arxiv_id":"2102.04990","repositories_listed":1,"syntology":{"n":10,"n_ran":4,"n_constructed":1,"n_ran_checked":1,"n_instrument":3,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":10,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/sg2caps-revisiting-scene-graphs-for-image#ran","syntology_url":"https://syntology.ai/paper/2102.04990","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.04990"}},"official":{"repos":["kien085/sg2caps"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":6,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/iconographic-image-captioning-for-artworks","slug":"iconographic-image-captioning-for-artworks","title":"Iconographic Image Captioning for Artworks","date":"2021-02-07","arxiv_id":"2102.03942","repositories_listed":1,"syntology":null},{"url":"/paper/the-role-of-syntactic-planning-in","slug":"the-role-of-syntactic-planning-in","title":"The Role of Syntactic Planning in Compositional Image Captioning","date":"2021-01-28","arxiv_id":"2101.11911","repositories_listed":1,"syntology":null},{"url":"/paper/dual-level-collaborative-transformer-for","slug":"dual-level-collaborative-transformer-for","title":"Dual-Level Collaborative Transformer for Image Captioning","date":"2021-01-16","arxiv_id":"2101.06462","repositories_listed":1,"syntology":null},{"url":"/paper/self-distillation-for-few-shot-image","slug":"self-distillation-for-few-shot-image","title":"Self-Distillation for Few-Shot Image Captioning","date":"2021-01-06","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/discovering-autoregressive-orderings-with","slug":"discovering-autoregressive-orderings-with","title":"Discovering Autoregressive Orderings with Variational Inference","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/text-free-image-to-speech-synthesis-using","slug":"text-free-image-to-speech-synthesis-using","title":"Text-Free Image-to-Speech Synthesis Using Learned Segmental Units","date":"2020-12-31","arxiv_id":"2012.15454","repositories_listed":1,"syntology":null},{"url":"/paper/detecting-hate-speech-in-multi-modal-memes","slug":"detecting-hate-speech-in-multi-modal-memes","title":"Detecting Hate Speech in Multi-modal Memes","date":"2020-12-29","arxiv_id":"2012.14891","repositories_listed":1,"syntology":null},{"url":"/paper/image-captioning-as-an-assistive-technology","slug":"image-captioning-as-an-assistive-technology","title":"Image Captioning as an Assistive Technology: Lessons Learned from VizWiz 2020 Challenge","date":"2020-12-21","arxiv_id":"2012.11696","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-cnn-lstm-based-image-captioning","slug":"efficient-cnn-lstm-based-image-captioning","title":"Efficient CNN-LSTM based Image Captioning using Neural Network Compression","date":"2020-12-17","arxiv_id":"2012.09708","repositories_listed":1,"syntology":null},{"url":"/paper/improving-image-captioning-by-leveraging-1","slug":"improving-image-captioning-by-leveraging-1","title":"Improving Image Captioning by Leveraging Intra- and Inter-layer Global Representation in Transformer Network","date":"2020-12-13","arxiv_id":"2012.07061","repositories_listed":1,"syntology":null},{"url":"/paper/simple-is-not-easy-a-simple-strong-baseline","slug":"simple-is-not-easy-a-simple-strong-baseline","title":"Simple is not Easy: A Simple Strong Baseline for TextVQA and TextCaps","date":"2020-12-09","arxiv_id":"2012.05153","repositories_listed":1,"syntology":null},{"url":"/paper/confidence-aware-non-repetitive-multimodal","slug":"confidence-aware-non-repetitive-multimodal","title":"Confidence-aware Non-repetitive Multimodal Transformers for TextCaps","date":"2020-12-07","arxiv_id":"2012.03662","repositories_listed":1,"syntology":null},{"url":"/paper/understanding-guided-image-captioning","slug":"understanding-guided-image-captioning","title":"Understanding Guided Image Captioning Performance across Domains","date":"2020-12-04","arxiv_id":"2012.02339","repositories_listed":1,"syntology":null}],"record_sha256":"1d5b63078b167424b814b2febb024807143127d22004e39214ad95b163fd4666","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}