{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/image-captioning/papers/8","list_of":"/task/image-captioning","task":"Image Captioning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":8,"pages_in_order":19,"rows_per_page":100,"rows":[701,800],"of":1878,"counts":{"archive_papers_tagged":1878,"with_a_code_link":774,"where_syntology_ran_a_sample":243,"not_listed_spam_title":0,"listed":1878,"listed_where_code_ran":243,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":201,"every_run_a_failure_of_syntologys_instrument":42,"listed_with_a_run_with_no_instrument_failure":201,"listed_every_run_a_failure_of_syntologys_instrument":42,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/image-captioning","prev":"/task/image-captioning/papers/7","next":"/task/image-captioning/papers/9","papers":[{"url":"/paper/a-neural-compositional-paradigm-for-image","slug":"a-neural-compositional-paradigm-for-image","title":"A Neural Compositional Paradigm for Image Captioning","date":"2018-10-23","arxiv_id":"1810.09630","repositories_listed":1,"syntology":null},{"url":"/paper/area-attention","slug":"area-attention","title":"Area Attention","date":"2018-10-23","arxiv_id":"1810.10126","repositories_listed":1,"syntology":null},{"url":"/paper/umons-submission-for-wmt18-multimodal","slug":"umons-submission-for-wmt18-multimodal","title":"UMONS Submission for WMT18 Multimodal Translation Task","date":"2018-10-15","arxiv_id":"1810.06233","repositories_listed":1,"syntology":null},{"url":"/paper/quantifying-the-amount-of-visual-information","slug":"quantifying-the-amount-of-visual-information","title":"Quantifying the amount of visual information used by neural caption generators","date":"2018-10-12","arxiv_id":"1810.05475","repositories_listed":1,"syntology":null},{"url":"/paper/image-captioning-as-neural-machine","slug":"image-captioning-as-neural-machine","title":"Image Captioning as Neural Machine Translation Task in SOCKEYE","date":"2018-10-09","arxiv_id":"1810.04101","repositories_listed":1,"syntology":null},{"url":"/paper/surprisingly-easy-hard-attention-for-sequence","slug":"surprisingly-easy-hard-attention-for-sequence","title":"Surprisingly Easy Hard-Attention for Sequence to Sequence Learning","date":"2018-10-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/training-for-diversity-in-image-paragraph","slug":"training-for-diversity-in-image-paragraph","title":"Training for Diversity in Image Paragraph Captioning","date":"2018-10-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/batch-normalized-recurrent-highway-networks","slug":"batch-normalized-recurrent-highway-networks","title":"Batch-normalized Recurrent Highway Networks","date":"2018-09-26","arxiv_id":"1809.10271","repositories_listed":1,"syntology":null},{"url":"/paper/fast-and-simple-mixture-of-softmaxes-with-bpe","slug":"fast-and-simple-mixture-of-softmaxes-with-bpe","title":"Fast and Simple Mixture of Softmaxes with BPE and Hybrid-LightRNN for Language Generation","date":"2018-09-25","arxiv_id":"1809.09296","repositories_listed":1,"syntology":null},{"url":"/paper/improving-reinforcement-learning-based-image","slug":"improving-reinforcement-learning-based-image","title":"Improving Reinforcement Learning Based Image Captioning with Natural Language Prior","date":"2018-09-13","arxiv_id":"1809.06227","repositories_listed":1,"syntology":null},{"url":"/paper/object-hallucination-in-image-captioning","slug":"object-hallucination-in-image-captioning","title":"Object Hallucination in Image Captioning","date":"2018-09-06","arxiv_id":"1809.02156","repositories_listed":1,"syntology":null},{"url":"/paper/accelerated-reinforcement-learning-for","slug":"accelerated-reinforcement-learning-for","title":"Accelerated Reinforcement Learning for Sentence Generation by Vocabulary Prediction","date":"2018-09-05","arxiv_id":"1809.01694","repositories_listed":1,"syntology":null},{"url":"/paper/simnet-stepwise-image-topic-merging-network","slug":"simnet-stepwise-image-topic-merging-network","title":"simNet: Stepwise Image-Topic Merging Network for Generating Detailed and Comprehensive Image Captions","date":"2018-08-27","arxiv_id":"1808.08732","repositories_listed":1,"syntology":null},{"url":"/paper/context-aware-visual-policy-network-for","slug":"context-aware-visual-policy-network-for","title":"Context-Aware Visual Policy Network for Sequence-Level Image Captioning","date":"2018-08-16","arxiv_id":"1808.05864","repositories_listed":1,"syntology":null},{"url":"/paper/topic-guided-attention-for-image-captioning","slug":"topic-guided-attention-for-image-captioning","title":"Topic-Guided Attention for Image Captioning","date":"2018-07-10","arxiv_id":"1807.03514","repositories_listed":1,"syntology":null},{"url":"/paper/face-cap-image-captioning-using-facial","slug":"face-cap-image-captioning-using-facial","title":"Face-Cap: Image Captioning using Facial Expression Analysis","date":"2018-07-06","arxiv_id":"1807.02250","repositories_listed":1,"syntology":null},{"url":"/paper/document-modeling-with-external-attention-for","slug":"document-modeling-with-external-attention-for","title":"Document Modeling with External Attention for Sentence Extraction","date":"2018-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-visually-grounded-semantics-from","slug":"learning-visually-grounded-semantics-from","title":"Learning Visually-Grounded Semantics from Contrastive Adversarial Samples","date":"2018-06-27","arxiv_id":"1806.10348","repositories_listed":1,"syntology":{"n":15,"n_ran":14,"n_constructed":0,"n_ran_checked":12,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":1,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-visually-grounded-semantics-from#ran","syntology_url":"https://syntology.ai/paper/1806.10348","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.10348"}},"official":{"repos":["ExplorerFreda/VSE-C"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-evaluate-image-captioning","slug":"learning-to-evaluate-image-captioning","title":"Learning to Evaluate Image Captioning","date":"2018-06-17","arxiv_id":"1806.06422","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/learning-to-evaluate-image-captioning#ran","syntology_url":"https://syntology.ai/paper/1806.06422","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.06422"}},"official":{"repos":["richardaecn/cvpr18-caption-eval"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/iparaphrasing-extracting-visually-grounded","slug":"iparaphrasing-extracting-visually-grounded","title":"iParaphrasing: Extracting Visually Grounded Paraphrases via an Image","date":"2018-06-12","arxiv_id":"1806.04284","repositories_listed":1,"syntology":null},{"url":"/paper/how-time-matters-learning-time-decay","slug":"how-time-matters-learning-time-decay","title":"How Time Matters: Learning Time-Decay Attention for Contextual Spoken Language Understanding in Dialogues","date":"2018-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/cnncnn-convolutional-decoders-for-image","slug":"cnncnn-convolutional-decoders-for-image","title":"CNN+CNN: Convolutional Decoders for Image Captioning","date":"2018-05-23","arxiv_id":"1805.09019","repositories_listed":1,"syntology":null},{"url":"/paper/improving-image-captioning-with-conditional","slug":"improving-image-captioning-with-conditional","title":"Improving Image Captioning with Conditional Generative Adversarial Nets","date":"2018-05-18","arxiv_id":"1805.07112","repositories_listed":1,"syntology":null},{"url":"/paper/semstyle-learning-to-generate-stylised-image","slug":"semstyle-learning-to-generate-stylised-image","title":"SemStyle: Learning to Generate Stylised Image Captions using Unaligned Text","date":"2018-05-18","arxiv_id":"1805.07030","repositories_listed":1,"syntology":null},{"url":"/paper/defoiling-foiled-image-captions","slug":"defoiling-foiled-image-captions","title":"Defoiling Foiled Image Captions","date":"2018-05-16","arxiv_id":"1805.06549","repositories_listed":1,"syntology":null},{"url":"/paper/token-level-and-sequence-level-loss-smoothing","slug":"token-level-and-sequence-level-loss-smoothing","title":"Token-level and sequence-level loss smoothing for RNN language models","date":"2018-05-14","arxiv_id":"1805.05062","repositories_listed":1,"syntology":null},{"url":"/paper/image-captioning","slug":"image-captioning","title":"Image Captioning","date":"2018-05-13","arxiv_id":"1805.09137","repositories_listed":1,"syntology":null},{"url":"/paper/visual-choice-of-plausible-alternatives-an","slug":"visual-choice-of-plausible-alternatives-an","title":"Visual Choice of Plausible Alternatives: An Evaluation of Image-based Commonsense Causal Reasoning","date":"2018-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/quantifying-the-visual-concreteness-of-words","slug":"quantifying-the-visual-concreteness-of-words","title":"Quantifying the visual concreteness of words and topics in multimodal datasets","date":"2018-04-18","arxiv_id":"1804.06786","repositories_listed":1,"syntology":null},{"url":"/paper/decoupled-novel-object-captioner","slug":"decoupled-novel-object-captioner","title":"Decoupled Novel Object Captioner","date":"2018-04-11","arxiv_id":"1804.03803","repositories_listed":1,"syntology":null},{"url":"/paper/finding-beans-in-burgers-deep-semantic-visual","slug":"finding-beans-in-burgers-deep-semantic-visual","title":"Finding beans in burgers: Deep semantic-visual embedding with localization","date":"2018-04-05","arxiv_id":"1804.01720","repositories_listed":1,"syntology":null},{"url":"/paper/regularizing-rnns-for-caption-generation-by","slug":"regularizing-rnns-for-caption-generation-by","title":"Regularizing RNNs for Caption Generation by Reconstructing The Past with The Present","date":"2018-03-30","arxiv_id":"1803.11439","repositories_listed":1,"syntology":null},{"url":"/paper/neural-baby-talk","slug":"neural-baby-talk","title":"Neural Baby Talk","date":"2018-03-27","arxiv_id":"1803.09845","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/neural-baby-talk#ran","syntology_url":"https://syntology.ai/paper/1803.09845","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1803.09845"}},"official":{"repos":["jiasenlu/NeuralBabyTalk"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/discriminability-objective-for-training","slug":"discriminability-objective-for-training","title":"Discriminability objective for training descriptive captions","date":"2018-03-12","arxiv_id":"1803.04376","repositories_listed":1,"syntology":null},{"url":"/paper/image-captioning-using-deep-neural","slug":"image-captioning-using-deep-neural","title":"Image Captioning using Deep Neural Architectures","date":"2018-01-17","arxiv_id":"1801.05568","repositories_listed":1,"syntology":null},{"url":"/paper/what-is-image-captioning-made-of","slug":"what-is-image-captioning-made-of","title":"What is image captioning made of?","date":"2018-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/fraternal-dropout","slug":"fraternal-dropout","title":"Fraternal Dropout","date":"2017-10-31","arxiv_id":"1711.00066","repositories_listed":1,"syntology":null},{"url":"/paper/cold-start-reinforcement-learning-with","slug":"cold-start-reinforcement-learning-with","title":"Cold-Start Reinforcement Learning with Softmax Policy Gradient","date":"2017-09-27","arxiv_id":"1709.09346","repositories_listed":1,"syntology":null},{"url":"/paper/stack-captioning-coarse-to-fine-learning-for","slug":"stack-captioning-coarse-to-fine-learning-for","title":"Stack-Captioning: Coarse-to-Fine Learning for Image Captioning","date":"2017-09-11","arxiv_id":"1709.03376","repositories_listed":1,"syntology":null},{"url":"/paper/convnet-architecture-search-for","slug":"convnet-architecture-search-for","title":"ConvNet Architecture Search for Spatiotemporal Feature Learning","date":"2017-08-16","arxiv_id":"1708.05038","repositories_listed":1,"syntology":null},{"url":"/paper/fluency-guided-cross-lingual-image-captioning","slug":"fluency-guided-cross-lingual-image-captioning","title":"Fluency-Guided Cross-Lingual Image Captioning","date":"2017-08-15","arxiv_id":"1708.04390","repositories_listed":1,"syntology":null},{"url":"/paper/udl-at-semeval-2017-task-1-semantic-textual","slug":"udl-at-semeval-2017-task-1-semantic-textual","title":"UdL at SemEval-2017 Task 1: Semantic Textual Similarity Estimation of English Sentence Pairs Using Regression Model over Pairwise Features","date":"2017-08-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/one-model-to-learn-them-all","slug":"one-model-to-learn-them-all","title":"One Model To Learn Them All","date":"2017-06-16","arxiv_id":"1706.05137","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/one-model-to-learn-them-all#ran","syntology_url":"https://syntology.ai/paper/1706.05137","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1706.05137"}},"official":{"repos":["tensorflow/tensor2tensor"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/show-adapt-and-tell-adversarial-training-of","slug":"show-adapt-and-tell-adversarial-training-of","title":"Show, Adapt and Tell: Adversarial Training of Cross-domain Image Captioner","date":"2017-05-02","arxiv_id":"1705.00930","repositories_listed":1,"syntology":null},{"url":"/paper/stair-captions-constructing-a-large-scale","slug":"stair-captions-constructing-a-large-scale","title":"STAIR Captions: Constructing a Large-Scale Japanese Image Caption Dataset","date":"2017-05-02","arxiv_id":"1705.00823","repositories_listed":1,"syntology":null},{"url":"/paper/towards-diverse-and-natural-image","slug":"towards-diverse-and-natural-image","title":"Towards Diverse and Natural Image Descriptions via a Conditional GAN","date":"2017-03-17","arxiv_id":"1703.06029","repositories_listed":1,"syntology":null},{"url":"/paper/context-aware-captions-from-context-agnostic","slug":"context-aware-captions-from-context-agnostic","title":"Context-aware Captions from Context-agnostic Supervision","date":"2017-01-11","arxiv_id":"1701.02870","repositories_listed":1,"syntology":null},{"url":"/paper/text-guided-attention-model-for-image","slug":"text-guided-attention-model-for-image","title":"Text-guided Attention Model for Image Captioning","date":"2016-12-12","arxiv_id":"1612.03557","repositories_listed":1,"syntology":null},{"url":"/paper/guided-open-vocabulary-image-captioning-with","slug":"guided-open-vocabulary-image-captioning-with","title":"Guided Open Vocabulary Image Captioning with Constrained Beam Search","date":"2016-12-02","arxiv_id":"1612.00576","repositories_listed":1,"syntology":null},{"url":"/paper/plug-play-generative-networks-conditional","slug":"plug-play-generative-networks-conditional","title":"Plug & Play Generative Networks: Conditional Iterative Generation of Images in Latent Space","date":"2016-11-30","arxiv_id":"1612.00005","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/plug-play-generative-networks-conditional#ran","syntology_url":"https://syntology.ai/paper/1612.00005","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1612.00005"}},"official":null}},{"url":"/paper/semantic-compositional-networks-for-visual","slug":"semantic-compositional-networks-for-visual","title":"Semantic Compositional Networks for Visual Captioning","date":"2016-11-23","arxiv_id":"1611.08002","repositories_listed":1,"syntology":null},{"url":"/paper/a-semi-supervised-framework-for-image","slug":"a-semi-supervised-framework-for-image","title":"A Semi-supervised Framework for Image Captioning","date":"2016-11-16","arxiv_id":"1611.05321","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-attention-for-neural-machine","slug":"multimodal-attention-for-neural-machine","title":"Multimodal Attention for Neural Machine Translation","date":"2016-09-13","arxiv_id":"1609.03976","repositories_listed":1,"syntology":null},{"url":"/paper/deepdiary-automatic-caption-generation-for","slug":"deepdiary-automatic-caption-generation-for","title":"DeepDiary: Automatic Caption Generation for Lifelogging Image Streams","date":"2016-08-12","arxiv_id":"1608.03819","repositories_listed":1,"syntology":null},{"url":"/paper/watch-what-you-just-said-image-captioning","slug":"watch-what-you-just-said-image-captioning","title":"Watch What You Just Said: Image Captioning with Text-Conditional Attention","date":"2016-06-15","arxiv_id":"1606.04621","repositories_listed":1,"syntology":null},{"url":"/paper/does-multimodality-help-human-and-machine-for","slug":"does-multimodality-help-human-and-machine-for","title":"Does Multimodality Help Human and Machine for Translation and Image Captioning?","date":"2016-05-30","arxiv_id":"1605.09186","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-instance-segmentation-with","slug":"end-to-end-instance-segmentation-with","title":"End-to-End Instance Segmentation with Recurrent Attention","date":"2016-05-30","arxiv_id":"1605.09410","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/end-to-end-instance-segmentation-with#ran","syntology_url":"https://syntology.ai/paper/1605.09410","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1605.09410"}},"official":null}},{"url":"/paper/annotation-order-matters-recurrent-image","slug":"annotation-order-matters-recurrent-image","title":"Annotation Order Matters: Recurrent Image Annotator for Arbitrary Length Image Tagging","date":"2016-04-18","arxiv_id":"1604.05225","repositories_listed":1,"syntology":null},{"url":"/paper/tgif-a-new-dataset-and-benchmark-on-animated","slug":"tgif-a-new-dataset-and-benchmark-on-animated","title":"TGIF: A New Dataset and Benchmark on Animated GIF Description","date":"2016-04-10","arxiv_id":"1604.02748","repositories_listed":1,"syntology":null},{"url":"/paper/image-captioning-with-deep-bidirectional","slug":"image-captioning-with-deep-bidirectional","title":"Image Captioning with Deep Bidirectional LSTMs","date":"2016-04-04","arxiv_id":"1604.00790","repositories_listed":1,"syntology":null},{"url":"/paper/densecap-fully-convolutional-localization","slug":"densecap-fully-convolutional-localization","title":"DenseCap: Fully Convolutional Localization Networks for Dense Captioning","date":"2015-11-24","arxiv_id":"1511.07571","repositories_listed":1,"syntology":null},{"url":"/paper/ask-attend-and-answer-exploring-question","slug":"ask-attend-and-answer-exploring-question","title":"Ask, Attend and Answer: Exploring Question-Guided Spatial Attention for Visual Question Answering","date":"2015-11-17","arxiv_id":"1511.05234","repositories_listed":1,"syntology":null},{"url":"/paper/deep-compositional-captioning-describing","slug":"deep-compositional-captioning-describing","title":"Deep Compositional Captioning: Describing Novel Object Categories without Paired Training Data","date":"2015-11-17","arxiv_id":"1511.05284","repositories_listed":1,"syntology":null},{"url":"/paper/how-not-to-train-your-generative-model","slug":"how-not-to-train-your-generative-model","title":"How (not) to Train your Generative Model: Scheduled Sampling, Likelihood, Adversary?","date":"2015-11-16","arxiv_id":"1511.05101","repositories_listed":1,"syntology":null},{"url":"/paper/oracle-performance-for-visual-captioning","slug":"oracle-performance-for-visual-captioning","title":"Oracle performance for visual captioning","date":"2015-11-14","arxiv_id":"1511.04590","repositories_listed":1,"syntology":null},{"url":"/paper/natural-language-object-retrieval","slug":"natural-language-object-retrieval","title":"Natural Language Object Retrieval","date":"2015-11-13","arxiv_id":"1511.04164","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/natural-language-object-retrieval#ran","syntology_url":"https://syntology.ai/paper/1511.04164","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1511.04164"}},"official":null}},{"url":"/paper/generation-and-comprehension-of-unambiguous","slug":"generation-and-comprehension-of-unambiguous","title":"Generation and Comprehension of Unambiguous Object Descriptions","date":"2015-11-07","arxiv_id":"1511.02283","repositories_listed":1,"syntology":null},{"url":"/paper/multilingual-image-description-with-neural","slug":"multilingual-image-description-with-neural","title":"Multilingual Image Description with Neural Sequence Models","date":"2015-10-15","arxiv_id":"1510.04709","repositories_listed":1,"syntology":null},{"url":"/paper/aligning-where-to-see-and-what-to-tell-image","slug":"aligning-where-to-see-and-what-to-tell-image","title":"Aligning where to see and what to tell: image caption with region-based attention and scene factorization","date":"2015-06-20","arxiv_id":"1506.06272","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-temporal-pooling-recurrence-and","slug":"beyond-temporal-pooling-recurrence-and","title":"Beyond Temporal Pooling: Recurrence and Temporal Convolutions for Gesture Recognition in Video","date":"2015-06-05","arxiv_id":"1506.01911","repositories_listed":1,"syntology":null},{"url":"/paper/what-value-do-explicit-high-level-concepts","slug":"what-value-do-explicit-high-level-concepts","title":"What value do explicit high level concepts have in vision to language problems?","date":"2015-06-03","arxiv_id":"1506.01144","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-nearest-neighbor-approaches-for","slug":"exploring-nearest-neighbor-approaches-for","title":"Exploring Nearest Neighbor Approaches for Image Captioning","date":"2015-05-17","arxiv_id":"1505.04467","repositories_listed":1,"syntology":null},{"url":"/paper/learning-like-a-child-fast-novel-visual","slug":"learning-like-a-child-fast-novel-visual","title":"Learning like a Child: Fast Novel Visual Concept Learning from Sentence Descriptions of Images","date":"2015-04-25","arxiv_id":"1504.06692","repositories_listed":1,"syntology":null},{"url":"/paper/from-captions-to-visual-concepts-and-back","slug":"from-captions-to-visual-concepts-and-back","title":"From Captions to Visual Concepts and Back","date":"2014-11-18","arxiv_id":"1411.4952","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/from-captions-to-visual-concepts-and-back#ran","syntology_url":"https://syntology.ai/paper/1411.4952","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1411.4952"}},"official":{"repos":["s-gupta/visual-concepts"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":null,"slug":"language-guided-contrastive-audio-visual","title":"Language-Guided Contrastive Audio-Visual Masked Autoencoder with Automatically Generated Audio-Visual-Text Triplets from Videos","date":"2025-07-16","arxiv_id":"2507.11967","repositories_listed":0,"syntology":null},{"url":null,"slug":"mask-aware-text-to-image-retrieval-referring","title":"Mask-aware Text-to-Image Retrieval: Referring Expression Segmentation Meets Cross-modal Retrieval","date":"2025-06-28","arxiv_id":"2506.22864","repositories_listed":0,"syntology":null},{"url":null,"slug":"halloc-token-level-localization-of-1","title":"HalLoc: Token-level Localization of Hallucinations for Vision Language Models","date":"2025-06-12","arxiv_id":"2506.10286","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-novel-lightweight-transformer-with-edge","title":"A Novel Lightweight Transformer with Edge-Aware Fusion for Remote Sensing Image Captioning","date":"2025-06-11","arxiv_id":"2506.09429","repositories_listed":0,"syntology":null},{"url":"/paper/an-open-source-software-toolkit-benchmark","slug":"an-open-source-software-toolkit-benchmark","title":"An Open-Source Software Toolkit & Benchmark Suite for the Evaluation and Adaptation of Multimodal Action Models","date":"2025-06-10","arxiv_id":"2506.09172","repositories_listed":0,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/an-open-source-software-toolkit-benchmark#ran","syntology_url":"https://syntology.ai/paper/2506.09172","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.09172"}},"official":null}},{"url":null,"slug":"better-reasoning-with-less-data-enhancing","title":"Better Reasoning with Less Data: Enhancing VLMs Through Unified Modality Scoring","date":"2025-06-10","arxiv_id":"2506.08429","repositories_listed":0,"syntology":null},{"url":null,"slug":"edit-flows-flow-matching-with-edit-operations","title":"Edit Flows: Flow Matching with Edit Operations","date":"2025-06-10","arxiv_id":"2506.09018","repositories_listed":0,"syntology":null},{"url":null,"slug":"2506-07553","title":"GTR-CoT: Graph Traversal as Visual Chain of Thought for Molecular Structure Recognition","date":"2025-06-09","arxiv_id":"2506.07553","repositories_listed":0,"syntology":null},{"url":null,"slug":"hallucination-at-a-glance-controlled-visual","title":"Hallucination at a Glance: Controlled Visual Edits and Fine-Grained Multimodal Learning","date":"2025-06-08","arxiv_id":"2506.07227","repositories_listed":0,"syntology":null},{"url":null,"slug":"srd-reinforcement-learned-semantic","title":"SRD: Reinforcement-Learned Semantic Perturbation for Backdoor Defense in VLMs","date":"2025-06-05","arxiv_id":"2506.04743","repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-based-transformer-models-for-image","title":"Attention-based transformer models for image captioning across languages: An in-depth survey and evaluation","date":"2025-06-03","arxiv_id":"2506.05399","repositories_listed":0,"syntology":null},{"url":null,"slug":"light-as-deception-gpt-driven-natural","title":"Light as Deception: GPT-driven Natural Relighting Against Vision-Language Pre-training Models","date":"2025-05-30","arxiv_id":"2505.24227","repositories_listed":0,"syntology":null},{"url":null,"slug":"beam-guided-knowledge-replay-for-knowledge","title":"Beam-Guided Knowledge Replay for Knowledge-Rich Image Captioning using Vision-Language Model","date":"2025-05-29","arxiv_id":"2505.23358","repositories_listed":0,"syntology":null},{"url":null,"slug":"tng-clip-training-time-negation-data","title":"TNG-CLIP:Training-Time Negation Data Generation for Negation Awareness of CLIP","date":"2025-05-24","arxiv_id":"2505.18434","repositories_listed":0,"syntology":null},{"url":null,"slug":"redemption-score-an-evaluation-framework-to","title":"Redemption Score: An Evaluation Framework to Rank Image Captions While Redeeming Image Semantics and Language Pragmatics","date":"2025-05-22","arxiv_id":"2505.16180","repositories_listed":0,"syntology":null},{"url":null,"slug":"steering-lvlms-via-sparse-autoencoder-for","title":"Steering LVLMs via Sparse Autoencoder for Hallucination Mitigation","date":"2025-05-22","arxiv_id":"2505.16146","repositories_listed":0,"syntology":null},{"url":null,"slug":"aligning-attention-distribution-to","title":"Aligning Attention Distribution to Information Flow for Hallucination Mitigation in Large Vision-Language Models","date":"2025-05-20","arxiv_id":"2505.14257","repositories_listed":0,"syntology":null},{"url":null,"slug":"medblip-fine-tuning-blip-for-medical-image","title":"MedBLIP: Fine-tuning BLIP for Medical Image Captioning","date":"2025-05-20","arxiv_id":"2505.14726","repositories_listed":0,"syntology":null},{"url":null,"slug":"nova-a-benchmark-for-anomaly-localization-and","title":"NOVA: A Benchmark for Anomaly Localization and Clinical Reasoning in Brain MRI","date":"2025-05-20","arxiv_id":"2505.14064","repositories_listed":0,"syntology":null},{"url":null,"slug":"sat2sound-a-unified-framework-for-zero-shot","title":"Sat2Sound: A Unified Framework for Zero-Shot Soundscape Mapping","date":"2025-05-19","arxiv_id":"2505.13777","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-10634","title":"Cross-Image Contrastive Decoding: Precise, Lossless Suppression of Language Priors in Large Vision-Language Models","date":"2025-05-15","arxiv_id":"2505.10634","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-grounded-memory-system-for-smart-personal","title":"A Grounded Memory System For Smart Personal Assistants","date":"2025-05-09","arxiv_id":"2505.06328","repositories_listed":0,"syntology":null},{"url":null,"slug":"artrag-retrieval-augmented-generation-with","title":"ArtRAG: Retrieval-Augmented Generation with Structured Context for Visual Art Understanding","date":"2025-05-09","arxiv_id":"2505.06020","repositories_listed":0,"syntology":null},{"url":null,"slug":"describe-anything-in-medical-images","title":"Describe Anything in Medical Images","date":"2025-05-09","arxiv_id":"2505.05804","repositories_listed":0,"syntology":null},{"url":null,"slug":"mitigating-image-captioning-hallucinations-in","title":"Mitigating Image Captioning Hallucinations in Vision-Language Models","date":"2025-05-06","arxiv_id":"2505.03420","repositories_listed":0,"syntology":null},{"url":null,"slug":"compositional-image-text-matching-and","title":"Compositional Image-Text Matching and Retrieval by Grounding Entities","date":"2025-05-04","arxiv_id":"2505.02278","repositories_listed":0,"syntology":null}],"record_sha256":"71afe9250796b58a3ce58604f590c8f5989508a834c07ac799ee73ef80ce7311","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}