{"url":"/task/visual-storytelling","name":"Visual Storytelling","slug":"visual-storytelling","description_markdown":"<span style=\"color:grey; opacity: 0.6\">( Image credit: [No Metrics Are Perfect](https://github.com/eric-xw/AREL) )</span>","categories":[{"name":"Natural Language Processing","url":"/area/natural-language-processing"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":115,"papers_with_code":37,"benchmarks":1,"benchmark_tables_in_archive":1,"benchmark_tables_shown":1,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":4,"subtasks":1,"parent_tasks":2},"benchmarks":[{"leaderboard":"/sota/visual-storytelling-on-vist","slug":"visual-storytelling-on-vist","dataset":"VIST","dataset_url":"/dataset/vist","rows_in_archive":33,"metrics":["BLEU-4","CIDEr","METEOR","BLEU-1","BLEU-2","BLEU-3","ROUGE-L","SPICE","BLEURT","MLTD"],"first_row_in_archive_order":{"model":"HEGR","paper_title":"Two Heads are Better Than One: Hypergraph-Enhanced Graph Reasoning for Visual Event Ratiocination","paper_url":"/paper/two-heads-are-better-than-one-hypergraph","paper_date":"2021-07-18","arxiv_id":null,"code_links":[],"syntology":null}}],"datasets":[{"url":"/dataset/vist","name":"VIST","full_name":"Visual Storytelling","num_papers_in_archive":107},{"url":"/dataset/visual-writing-prompts","name":"Visual Writing Prompts","full_name":"","num_papers_in_archive":3},{"url":"/dataset/vist-edit","name":"VIST-Edit","full_name":"VIST-Edit","num_papers_in_archive":2},{"url":"/dataset/creative-visual-storytelling-anthology","name":"Creative Visual Storytelling Anthology","full_name":"ARL Creative Visual Storytelling Anthology","num_papers_in_archive":1}],"subtasks":[{"url":"/task/image-guided-story-ending-generation","name":"Image-guided Story Ending Generation"}],"parent_tasks":[{"url":"/task/data-to-text-generation","name":"Data-to-Text Generation"},{"url":"/task/story-generation","name":"Story Generation"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":37,"tagged_in_all":115,"items":[{"url":"/paper/alfie-democratising-rgba-image-generation","title":"Alfie: Democratising RGBA Image Generation With No $$$","date":"2024-08-27","arxiv_id":"2408.14826","repositories_listed":2,"syntology":null},{"url":"/paper/aesop-abstract-encoding-of-stories-objects","title":"AESOP: Abstract Encoding of Stories, Objects, and Pictures","date":"2021-01-01","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/contextualize-show-and-tell-a-neural-visual","title":"Contextualize, Show and Tell: A Neural Visual Storyteller","date":"2018-06-03","arxiv_id":"1806.00738","repositories_listed":2,"syntology":null},{"url":"/paper/no-metrics-are-perfect-adversarial-reward","title":"No Metrics Are Perfect: Adversarial Reward Learning for Visual Storytelling","date":"2018-04-24","arxiv_id":"1804.09160","repositories_listed":2,"syntology":{"n":8,"n_ran":1,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/consistent-story-generation-with-asymmetry","title":"Consistent Story Generation with Asymmetry Zigzag Sampling","date":"2025-06-11","arxiv_id":"2506.09612","repositories_listed":1,"syntology":null},{"url":"/paper/storyreasoning-dataset-using-chain-of-thought","title":"StoryReasoning Dataset: Using Chain-of-Thought for Scene Understanding and Grounded Story Generation","date":"2025-05-15","arxiv_id":"2505.10292","repositories_listed":1,"syntology":null},{"url":"/paper/flip-reasoning-challenge","title":"FLIP Reasoning Challenge","date":"2025-04-16","arxiv_id":"2504.12256","repositories_listed":1,"syntology":null},{"url":"/paper/flipsketch-flipping-static-drawings-to-text","title":"FlipSketch: Flipping Static Drawings to Text-Guided Sketch Animations","date":"2024-11-16","arxiv_id":"2411.10818","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-large-language-models-and-tunings","title":"Multimodal Large Language Models and Tunings: Vision, Language, Sensors, Audio, and Beyond","date":"2024-10-08","arxiv_id":"2410.05608","repositories_listed":1,"syntology":null},{"url":"/paper/openstory-a-large-scale-dataset-and-benchmark","title":"Openstory++: A Large-scale Dataset and Benchmark for Instance-aware Open-domain Visual Storytelling","date":"2024-08-07","arxiv_id":"2408.03695","repositories_listed":1,"syntology":null},{"url":"/paper/temporalstory-enhancing-consistency-in-story","title":"ContextualStory: Consistent Visual Storytelling with Spatially-Enhanced and Storyline Context","date":"2024-07-13","arxiv_id":"2407.09774","repositories_listed":1,"syntology":null},{"url":"/paper/not-yet-the-whole-story-evaluating-visual","title":"Not (yet) the whole story: Evaluating Visual Storytelling Requires More than Measuring Coherence, Grounding, and Repetition","date":"2024-07-05","arxiv_id":"2407.04559","repositories_listed":1,"syntology":null},{"url":"/paper/comm-a-coherent-interleaved-image-text","title":"CoMM: A Coherent Interleaved Image-Text Dataset for Multimodal Understanding and Generation","date":"2024-06-15","arxiv_id":"2406.10462","repositories_listed":1,"syntology":null},{"url":"/paper/gorgeous-create-your-desired-character-facial","title":"Gorgeous: Create Your Desired Character Facial Makeup from Any Ideas","date":"2024-04-22","arxiv_id":"2404.13944","repositories_listed":1,"syntology":null},{"url":"/paper/intelligent-grimm-open-ended-visual-1","title":"Intelligent Grimm - Open-ended Visual Storytelling via Latent Diffusion Models","date":"2024-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/inkn-hue-enhancing-manga-colorization-from","title":"inkn'hue: Enhancing Manga Colorization from Multiple Priors with Alignment Multi-Encoder VAE","date":"2023-11-03","arxiv_id":"2311.01804","repositories_listed":1,"syntology":null},{"url":"/paper/groovist-a-metric-for-grounding-objects-in","title":"GROOViST: A Metric for Grounding Objects in Visual Storytelling","date":"2023-10-26","arxiv_id":"2310.17770","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":3}},{"url":"/paper/envisioning-narrative-intelligence-a-creative","title":"Envisioning Narrative Intelligence: A Creative Visual Storytelling Anthology","date":"2023-10-06","arxiv_id":"2310.04529","repositories_listed":1,"syntology":null},{"url":"/paper/touchstone-evaluating-vision-language-models","title":"TouchStone: Evaluating Vision-Language Models by Language Models","date":"2023-08-31","arxiv_id":"2308.16890","repositories_listed":1,"syntology":null},{"url":"/paper/animate-a-story-storytelling-with-retrieval","title":"Animate-A-Story: Storytelling with Retrieval-Augmented Video Generation","date":"2023-07-13","arxiv_id":"2307.06940","repositories_listed":1,"syntology":null},{"url":"/paper/intelligent-grimm-open-ended-visual","title":"Intelligent Grimm -- Open-ended Visual Storytelling via Latent Diffusion Models","date":"2023-06-01","arxiv_id":"2306.00973","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/visual-transformation-telling","title":"Visual Transformation Telling","date":"2023-05-03","arxiv_id":"2305.01928","repositories_listed":1,"syntology":null},{"url":"/paper/detecting-and-grounding-important-characters","title":"Detecting and Grounding Important Characters in Visual Stories","date":"2023-03-30","arxiv_id":"2303.17647","repositories_listed":1,"syntology":null},{"url":"/paper/positional-diffusion-ordering-unordered-sets","title":"Positional Diffusion: Ordering Unordered Sets with Diffusion Probabilistic Models","date":"2023-03-20","arxiv_id":"2303.11120","repositories_listed":1,"syntology":null},{"url":"/paper/rovist-learning-robust-metrics-for-visual-2","title":"RoViST: Learning Robust Metrics for Visual Storytelling","date":"2022-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/expressive-scene-graph-generation-using","title":"Expressive Scene Graph Generation Using Commonsense Knowledge Infusion for Visual Understanding and Reasoning","date":"2022-05-31","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/rovist-learning-robust-metrics-for-visual-1","title":"RoViST:Learning Robust Metrics for Visual Storytelling","date":"2022-05-08","arxiv_id":"2205.03774","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-rank-visual-stories-from-human-1","title":"Learning to Rank Visual Stories From Human Ranking Data","date":"2022-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/plot-and-rework-modeling-storylines-for","title":"Plot and Rework: Modeling Storylines for Visual Storytelling","date":"2021-05-14","arxiv_id":"2105.06950","repositories_listed":1,"syntology":null},{"url":"/paper/knowledge-enriched-visual-storytelling","title":"Knowledge-Enriched Visual Storytelling","date":"2019-12-03","arxiv_id":"1912.01496","repositories_listed":1,"syntology":null}],"syntology_records":3,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}