{"url":"/sota/visual-storytelling-on-vist","task":{"name":"Visual Storytelling","url":"/task/visual-storytelling","note":null},"dataset":{"name":"VIST","url":"/dataset/vist"},"category":"Natural Language Processing","categories":["Natural Language Processing"],"category_note":null,"description":"<span style=\"color:grey; opacity: 0.6\">( Image credit: [No Metrics Are Perfect](https://github.com/eric-xw/AREL) )</span>","description_from":"task","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","rank":"the archive's row order at snapshot; not re-ranked","rows_end_at":"2025-07-28","rows_withheld_as_spam":0,"metric_values":"the archive's strings, untouched"},"metrics":["BLEU-4","CIDEr","METEOR","BLEU-1","BLEU-2","BLEU-3","ROUGE-L","SPICE","BLEURT","MLTD"],"metric_direction":{"note":"inferred from the metric name only (the archive records no direction); null = not inferred, chart draws points only","by_metric":{"BLEU-4":"higher","CIDEr":null,"METEOR":null,"BLEU-1":"higher","BLEU-2":"higher","BLEU-3":"higher","ROUGE-L":"higher","SPICE":null,"BLEURT":null,"MLTD":null}},"counts":{"rows":33,"rows_with_code":13,"rows_with_paper_page":33,"rows_dated":33,"rows_using_additional_data":2},"rows":[{"rank_in_archive_order":1,"model":"HEGR","metrics":{"BLEU-4":"16.7","CIDEr":"14.1","METEOR":"37.8"},"uses_additional_data":true,"paper_date":"2021-07-18","paper":"/paper/two-heads-are-better-than-one-hypergraph","paper_url":"https://proceedings.mlr.press/v139/zheng21b.html","paper_title":"Two Heads are Better Than One: Hypergraph-Enhanced Graph Reasoning for Visual Event Ratiocination","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":2,"model":"IRW","metrics":{"BLEU-1":"66.7","BLEU-2":"41.6","BLEU-3":"25.0","BLEU-4":"15.4","CIDEr":"11.0","METEOR":"35.6","ROUGE-L":"29.6"},"uses_additional_data":false,"paper_date":"2021-05-18","paper":"/paper/imagine-reason-and-write-visual-storytelling","paper_url":"https://ojs.aaai.org/index.php/AAAI/article/view/16410","paper_title":"Imagine, Reason and Write: Visual Storytelling with Graph Knowledge and Relational Reasoning","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":3,"model":"HBSG","metrics":{"BLEU-4":"15.4","METEOR":"36.5"},"uses_additional_data":false,"paper_date":"2022-01-10","paper":"/paper/visual-storytelling-with-hierarchical-bert","paper_url":"https://dl.acm.org/doi/abs/10.1145/3469877.3490604","paper_title":"Visual Storytelling with Hierarchical BERT Semantic Guidance","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":4,"model":"CoVS","metrics":{"BLEU-1":"67.5","BLEU-2":"42.7","BLEU-3":"25.3","BLEU-4":"15.2","CIDEr":"11.5","METEOR":"36.5","ROUGE-L":"30.8"},"uses_additional_data":false,"paper_date":"2022-08-17","paper":"/paper/coherent-visual-storytelling-via-parallel-top","paper_url":"https://ieeexplore.ieee.org/document/9858883","paper_title":"Coherent Visual Storytelling via Parallel Top-Down Visual and Topic Attention","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":5,"model":"SentiStory","metrics":{"BLEU-1":"65.5","BLEU-2":"40.7","BLEU-3":"24.1","BLEU-4":"14.8","CIDEr":"10.1","METEOR":"35.7","ROUGE-L":"30.2"},"uses_additional_data":false,"paper_date":"2022-06-16","paper":"/paper/sentistory-a-multi-layered-sentiment-aware","paper_url":"https://ieeexplore.ieee.org/document/9797749","paper_title":"SentiStory: A Multi-Layered Sentiment-Aware Generative Model for Visual Storytelling","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":6,"model":"SGEmb","metrics":{"BLEU-1":"62.2","BLEU-2":"38.7","BLEU-3":"23.5","BLEU-4":"14.8","CIDEr":"8.6","METEOR":"35.6","ROUGE-L":"30.2"},"uses_additional_data":false,"paper_date":"2020-11-01","paper":"/paper/diverse-and-relevant-visual-storytelling-with","paper_url":"https://aclanthology.org/2020.conll-1.34","paper_title":"Diverse and Relevant Visual Storytelling with Scene Graph Embeddings","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":7,"model":"INet","metrics":{"BLEU-1":"64.4","BLEU-2":"0.401","BLEU-3":"23.9","BLEU-4":"14.7","CIDEr":"10","METEOR":"35.6","ROUGE-L":"29.7"},"uses_additional_data":false,"paper_date":"2020-02-03","paper":"/paper/hide-and-tell-learning-to-bridge-photo","paper_url":"https://arxiv.org/abs/2002.00774v1","paper_title":"Hide-and-Tell: Learning to Bridge Photo Streams for Visual Storytelling","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":8,"model":"SGVST","metrics":{"BLEU-4":"14.7","CIDEr":"9.8","METEOR":"35.8","ROUGE-L":"29.9"},"uses_additional_data":false,"paper_date":"2020-04-03","paper":"/paper/storytelling-from-an-image-stream-using-scene","paper_url":"https://ojs.aaai.org/index.php/AAAI/article/view/6455","paper_title":"Storytelling from an Image Stream Using Scene Graphs","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":9,"model":"TAVST (RL)","metrics":{"BLEU-1":"64.2","BLEU-2":"39.6","BLEU-3":"23.7","BLEU-4":"14.6","CIDEr":"9.2","METEOR":"35.7","ROUGE-L":"31"},"uses_additional_data":false,"paper_date":"2019-11-11","paper":"/paper/keep-it-consistent-topic-aware-storytelling","paper_url":"https://arxiv.org/abs/1911.04192v2","paper_title":"Keep it Consistent: Topic-Aware Storytelling from an Image Stream via Iterative Multi-agent Communication","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":10,"model":"BLEU-RL","metrics":{"BLEU-4":"14.4","CIDEr":"6.7","METEOR":"35.2","ROUGE-L":"30.1","SPICE":"8.3"},"uses_additional_data":false,"paper_date":"2019-09-11","paper":"/paper/what-makes-a-good-story-designing-composite","paper_url":"https://arxiv.org/abs/1909.05316v2","paper_title":"What Makes A Good Story? Designing Composite Rewards for Visual Storytelling","code":"https://github.com/JunjieHu/ReCo-RL","n_code_links":1,"syntology":{"n_ran":0,"n_unverified":12,"n_samples":12,"n_pointer_only_licence":0}},{"rank_in_archive_order":11,"model":"VSCMR","metrics":{"BLEU-1":"63.8","BLEU-4":"14.3","CIDEr":"9","METEOR":"35.5","ROUGE-L":"30.2"},"uses_additional_data":false,"paper_date":"2019-07-07","paper":"/paper/informative-visual-storytelling-with-cross","paper_url":"https://arxiv.org/abs/1907.03240v2","paper_title":"Informative Visual Storytelling with Cross-modal Rules","code":"https://github.com/passerby233/VSCMR-Visual-Storytelling-with-Corss-Modal-Rules","n_code_links":1,"syntology":null},{"rank_in_archive_order":12,"model":"MLE","metrics":{"BLEU-4":"14.3","CIDEr":"7.2","METEOR":"34.8","ROUGE-L":"30","SPICE":"8.5"},"uses_additional_data":false,"paper_date":"2019-09-11","paper":"/paper/what-makes-a-good-story-designing-composite","paper_url":"https://arxiv.org/abs/1909.05316v2","paper_title":"What Makes A Good Story? Designing Composite Rewards for Visual Storytelling","code":"https://github.com/JunjieHu/ReCo-RL","n_code_links":1,"syntology":{"n_ran":0,"n_unverified":12,"n_samples":12,"n_pointer_only_licence":0}},{"rank_in_archive_order":13,"model":"AREL-t-100","metrics":{"BLEU-1":"63.8","BLEU-2":"39.1","BLEU-3":"23.2","BLEU-4":"14.1","CIDEr":"9.4","METEOR":"35","ROUGE-L":"29.5"},"uses_additional_data":false,"paper_date":"2018-04-24","paper":"/paper/no-metrics-are-perfect-adversarial-reward","paper_url":"http://arxiv.org/abs/1804.09160v2","paper_title":"No Metrics Are Perfect: Adversarial Reward Learning for Visual Storytelling","code":"https://github.com/eric-xw/AREL","n_code_links":2,"syntology":{"n_ran":1,"n_unverified":7,"n_samples":8,"n_pointer_only_licence":0}},{"rank_in_archive_order":14,"model":"MemNet","metrics":{"BLEU-4":"14.1","METEOR":"35.5"},"uses_additional_data":false,"paper_date":"2020-09-01","paper":"/paper/hierarchical-memory-decoder-for-visual","paper_url":"https://ieeexplore.ieee.org/abstract/document/9184025/authors","paper_title":"Hierarchical memory decoder for visual narrating","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":15,"model":"StoryAnchor: w/ Predicted Nouns","metrics":{"BLEU-1":"65.1","BLEU-2":"40.0","BLEU-3":"23.4","BLEU-4":"14","CIDEr":"9.9","METEOR":"35.5","ROUGE-L":"30"},"uses_additional_data":false,"paper_date":"2020-01-13","paper":"/paper/visual-storytelling-via-predicting-anchor","paper_url":"https://arxiv.org/abs/2001.04541v1","paper_title":"Visual Storytelling via Predicting Anchor Word Embeddings in the Stories","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":16,"model":"GAN","metrics":{"BLEU-1":"62.8","BLEU-2":"38.8","BLEU-3":"23.0","BLEU-4":"14","CIDEr":"9","METEOR":"35","ROUGE-L":"29.5"},"uses_additional_data":false,"paper_date":"2018-04-24","paper":"/paper/no-metrics-are-perfect-adversarial-reward","paper_url":"http://arxiv.org/abs/1804.09160v2","paper_title":"No Metrics Are Perfect: Adversarial Reward Learning for Visual Storytelling","code":"https://github.com/eric-xw/AREL","n_code_links":2,"syntology":{"n_ran":1,"n_unverified":7,"n_samples":8,"n_pointer_only_licence":0}},{"rank_in_archive_order":17,"model":"XE-ss","metrics":{"BLEU-1":"62.3","BLEU-2":"38.2","BLEU-3":"22.5","BLEU-4":"13.7","CIDEr":"8.7","METEOR":"34.8","ROUGE-L":"29.7"},"uses_additional_data":false,"paper_date":"2018-04-24","paper":"/paper/no-metrics-are-perfect-adversarial-reward","paper_url":"http://arxiv.org/abs/1804.09160v2","paper_title":"No Metrics Are Perfect: Adversarial Reward Learning for Visual Storytelling","code":"https://github.com/eric-xw/AREL","n_code_links":2,"syntology":{"n_ran":1,"n_unverified":7,"n_samples":8,"n_pointer_only_licence":0}},{"rank_in_archive_order":18,"model":"AREL","metrics":{"BLEU-4":"13.6","CIDEr":"9.1","METEOR":"35.2","ROUGE-L":"29.3","SPICE":"8.9"},"uses_additional_data":false,"paper_date":"2019-09-11","paper":"/paper/what-makes-a-good-story-designing-composite","paper_url":"https://arxiv.org/abs/1909.05316v2","paper_title":"What Makes A Good Story? Designing Composite Rewards for Visual Storytelling","code":"https://github.com/JunjieHu/ReCo-RL","n_code_links":1,"syntology":{"n_ran":0,"n_unverified":12,"n_samples":12,"n_pointer_only_licence":0}},{"rank_in_archive_order":19,"model":"MCSM+RNN","metrics":{"BLEU-3":"23.1","BLEU-4":"13","CIDEr":"11","METEOR":"36.1","ROUGE-L":"30.7"},"uses_additional_data":false,"paper_date":"2021-02-05","paper":"/paper/commonsense-knowledge-aware-concept-selection","paper_url":"https://arxiv.org/abs/2102.02963v1","paper_title":"Commonsense Knowledge Aware Concept Selection For Diverse and Informative Visual Storytelling","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":20,"model":"AOG + ARS","metrics":{"BLEU-1":"69","BLEU-2":"44","BLEU-3":"23.9","BLEU-4":"12.9","CIDEr":"12.0","METEOR":"36.0","ROUGE-L":"30.1"},"uses_additional_data":false,"paper_date":"2023-06-26","paper":"/paper/aog-lstm-an-adaptive-attention-neural-network","paper_url":"https://www.sciencedirect.com/science/article/pii/S0925231223006094","paper_title":"AOG-LSTM: An adaptive attention neural network for visual storytelling","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":21,"model":"K-Storyteller","metrics":{"BLEU-4":"12.8","CIDEr":"12.1","METEOR":"35.2","ROUGE-L":"29.9"},"uses_additional_data":false,"paper_date":"2019-05-04","paper":"/paper/knowledgeable-storyteller-a-commonsense","paper_url":"https://www.ijcai.org/proceedings/2019/744","paper_title":"Knowledgeable Storyteller: A Commonsense-Driven Generative Model for Visual Storytelling","code":"https://github.com/lancopku/CVST","n_code_links":1,"syntology":null},{"rank_in_archive_order":22,"model":"CST","metrics":{"BLEU-1":"60.1","BLEU-2":"36.5","BLEU-3":"21.1","BLEU-4":"12.7","CIDEr":"5.1","METEOR":"34.4","ROUGE-L":"29.2"},"uses_additional_data":false,"paper_date":"2018-06-03","paper":"/paper/contextualize-show-and-tell-a-neural-visual","paper_url":"http://arxiv.org/abs/1806.00738v1","paper_title":"Contextualize, Show and Tell: A Neural Visual Storyteller","code":"https://github.com/dgonzalez-ri/neural-visual-storyteller","n_code_links":2,"syntology":null},{"rank_in_archive_order":23,"model":"ReCo-RL","metrics":{"BLEU-4":"12.4","CIDEr":"8.6","METEOR":"33.9","ROUGE-L":"29.9","SPICE":"8.3"},"uses_additional_data":false,"paper_date":"2019-09-11","paper":"/paper/what-makes-a-good-story-designing-composite","paper_url":"https://arxiv.org/abs/1909.05316v2","paper_title":"What Makes A Good Story? Designing Composite Rewards for Visual Storytelling","code":"https://github.com/JunjieHu/ReCo-RL","n_code_links":1,"syntology":{"n_ran":0,"n_unverified":12,"n_samples":12,"n_pointer_only_licence":0}},{"rank_in_archive_order":24,"model":"HSRL w/ Joint Training","metrics":{"BLEU-4":"12.32","CIDEr":"10.71","METEOR":"35.23","ROUGE-L":"30.84","SPICE":"12.97"},"uses_additional_data":false,"paper_date":"2018-05-21","paper":"/paper/hierarchically-structured-reinforcement","paper_url":"http://arxiv.org/abs/1805.08191v3","paper_title":"Hierarchically Structured Reinforcement Learning for Topically Coherent Visual Story Generation","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":25,"model":"ViT-model","metrics":{"BLEU-1":"63","BLEU-2":"37.5","BLEU-3":"21.5","BLEU-4":"12.3","CIDEr":"4.4","METEOR":"35.4","ROUGE-L":"31"},"uses_additional_data":false,"paper_date":"2022-10-06","paper":"/paper/vision-transformer-based-model-for-describing","paper_url":"https://arxiv.org/abs/2210.02762v3","paper_title":"Vision Transformer Based Model for Describing a Set of Images as a Story","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":26,"model":"HSRL","metrics":{"BLEU-4":"9.8","CIDEr":"5.9","METEOR":"30.1","ROUGE-L":"25.1","SPICE":"7.5"},"uses_additional_data":false,"paper_date":"2019-09-11","paper":"/paper/what-makes-a-good-story-designing-composite","paper_url":"https://arxiv.org/abs/1909.05316v2","paper_title":"What Makes A Good Story? Designing Composite Rewards for Visual Storytelling","code":"https://github.com/JunjieHu/ReCo-RL","n_code_links":1,"syntology":{"n_ran":0,"n_unverified":12,"n_samples":12,"n_pointer_only_licence":0}},{"rank_in_archive_order":27,"model":"PR-VIST","metrics":{"BLEU-4":"7.65","BLEURT":"1.37","METEOR":"31.6","MLTD":"45.79"},"uses_additional_data":false,"paper_date":"2021-05-14","paper":"/paper/plot-and-rework-modeling-storylines-for","paper_url":"https://arxiv.org/abs/2105.06950v3","paper_title":"Plot and Rework: Modeling Storylines for Visual Storytelling","code":"https://github.com/ethan5437/PR-VIST","n_code_links":1,"syntology":null},{"rank_in_archive_order":28,"model":"TAPM","metrics":{"CIDEr":"13.8","METEOR":"37.2","ROUGE-L":"33.1"},"uses_additional_data":true,"paper_date":"2021-06-19","paper":"/paper/transitional-adaptation-of-pretrained-models","paper_url":"http://openaccess.thecvf.com//content/CVPR2021/html/Yu_Transitional_Adaptation_of_Pretrained_Models_for_Visual_Storytelling_CVPR_2021_paper.html","paper_title":"Transitional Adaptation of Pretrained Models for Visual Storytelling","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":29,"model":"BERT-hLSTMs","metrics":{"CIDEr":"8.37"},"uses_additional_data":false,"paper_date":"2020-12-03","paper":"/paper/bert-hlstms-bert-and-hierarchical-lstms-for","paper_url":"https://arxiv.org/abs/2012.02128v1","paper_title":"BERT-hLSTMs: BERT and Hierarchical LSTMs for Visual Storytelling","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":30,"model":"TAPM (no V&L)","metrics":{"CIDEr":"8.3","METEOR":"34.1","ROUGE-L":"30.2"},"uses_additional_data":false,"paper_date":"2021-06-19","paper":"/paper/transitional-adaptation-of-pretrained-models","paper_url":"http://openaccess.thecvf.com//content/CVPR2021/html/Yu_Transitional_Adaptation_of_Pretrained_Models_for_Visual_Storytelling_CVPR_2021_paper.html","paper_title":"Transitional Adaptation of Pretrained Models for Visual Storytelling","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":31,"model":"hLSTMs","metrics":{"CIDEr":"7.98"},"uses_additional_data":false,"paper_date":"2020-12-03","paper":"/paper/bert-hlstms-bert-and-hierarchical-lstms-for","paper_url":"https://arxiv.org/abs/2012.02128v1","paper_title":"BERT-hLSTMs: BERT and Hierarchical LSTMs for Visual Storytelling","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":32,"model":"h-attn-rank","metrics":{"BLEU-3":"20.78","CIDEr":"7.38","METEOR":"33.94","ROUGE-L":"29.82"},"uses_additional_data":false,"paper_date":"2017-08-09","paper":"/paper/hierarchically-attentive-rnn-for-album","paper_url":"http://arxiv.org/abs/1708.02977v1","paper_title":"Hierarchically-Attentive RNN for Album Summarization and Storytelling","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":33,"model":"GLAC Net","metrics":{"METEOR":"30.14"},"uses_additional_data":false,"paper_date":"2018-05-28","paper":"/paper/glac-net-glocal-attention-cascading-networks","paper_url":"http://arxiv.org/abs/1805.10973v3","paper_title":"GLAC Net: GLocal Attention Cascading Networks for Multi-image Cued Story Generation","code":"https://github.com/tkim-snu/GLACNet","n_code_links":1,"syntology":null}],"since_archive":{"claim":"Results that newer papers report for their own method, placed here by Syntology. A model pointed at the cell in the paper's own table; the number was read from that cell and checked against this leaderboard's metric, dataset, split and scale; an independent check that saw this leaderboard's other rows and every other leaderboard on the same dataset accepted it. Not reviewed by the paper's authors or by the archive's editors, and not ranked against the archive rows.","extraction_file_present":true,"measurement":{"test_papers":883,"papers_with_output":881,"judged_true":108,"judged":110,"wilson95_lower":0.9361,"measured_on":"2026-09-24","frozen_commit":"0e3de0df94"},"measurement_note":"blind adjudication of accepted entries on a held-out split of archive papers, rules frozen before the test","coverage":{"sentence":"Syntology has checked 6,885 of the 9,623 papers on this site that are newer than the archive; results from the others appear after they are checked.","complete":false,"papers_newer_than_archive":9623,"papers_checked":6885,"papers_extracted_not_yet_verified":0,"boards_without_verdict":2,"papers_not_yet_extracted":2737},"order":"newest first by month (arXiv date, else the arXiv-id month), then arXiv id descending","columns":[],"entries":[]},"syntology":{"read_at":"2026-09-25T09:33:49+00:00","claim":"Per row: N of M harvested code samples from that row's paper executed on a synthesized fixture; the other M-N are unverified. Not a reproduction of the row's number; not a correctness claim. n_pointer_only_licence counts samples the site points at rather than redistributes (a licence axis, independent of ran/unverified).","rows_with_graph_line":8,"rows_with_any_sample_ran":3,"distinct_papers_with_graph_line":2,"distinct_papers_with_any_sample_ran":1,"samples_over_distinct_papers":{"n_ran":1,"n_unverified":19,"n_samples":20,"n_pointer_only_licence":0,"note":"each paper (arXiv id) counted once, however many rows it is behind; this is the page-level figure"},"samples_row_weighted":{"n_ran":3,"n_unverified":81,"n_samples":84,"n_pointer_only_licence":0,"note":"row-weighted: a paper behind several rows is counted once per row; inflated relative to samples_over_distinct_papers by design, kept for readers summing the per-row syntology blocks"}}}