{"url":"/task/story-visualization","name":"Story Visualization","slug":"story-visualization","description_markdown":"Story Visualization is the task of generating  coherent and aligned sequence of images given a sequence of textual captions representing description of a story.  It mainly consists of two tasks: story generation and story continuation, where story continuation uses additional ground truth information in the form of the first frame.","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":42,"papers_with_code":28,"benchmarks":3,"benchmark_tables_in_archive":3,"benchmark_tables_shown":3,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":1,"subtasks":0,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/story-visualization-on-pororo","slug":"story-visualization-on-pororo","dataset":"Pororo","dataset_url":null,"rows_in_archive":5,"metrics":["FID","FSD"],"first_row_in_archive_order":{"model":"ContextualStory","paper_title":"ContextualStory: Consistent Visual Storytelling with Spatially-Enhanced and Storyline Context","paper_url":"/paper/temporalstory-enhancing-consistency-in-story","paper_date":"2024-07-13","arxiv_id":"2407.09774","code_links":[{"title":"sixiaozheng/contextualstory","url":"https://github.com/sixiaozheng/contextualstory"}],"syntology":null}},{"leaderboard":"/sota/story-visualization-on-clevr-sv","slug":"story-visualization-on-clevr-sv","dataset":"CLEVR-SV","dataset_url":null,"rows_in_archive":1,"metrics":["LPIPS"],"first_row_in_archive_order":{"model":"Impartial Transformer","paper_title":"An Impartial Transformer for Story Visualization","paper_url":"/paper/an-impartial-transformer-for-story","paper_date":"2023-01-09","arxiv_id":"2301.03563","code_links":[],"syntology":null}},{"leaderboard":"/sota/story-visualization-on-zero-shot-action","slug":"story-visualization-on-zero-shot-action","dataset":"Zero-Shot Action Execution DiDeMO-CSV","dataset_url":"/dataset/storybench","rows_in_archive":1,"metrics":["FID_Iv3","FVD_I3D","SIM_Iv3","PQA","VTM_CLIP","FID_CLIP","FVD_IV","SIM_CLIP","VTM_IV"],"first_row_in_archive_order":{"model":"Phenaki-Gen","paper_title":null,"paper_url":null,"paper_date":"","arxiv_id":null,"code_links":[],"syntology":null}}],"datasets":[{"url":"/dataset/storybench","name":"StoryBench","full_name":"StoryBench: A Multifaceted Benchmark for Continuous Story Visualization","num_papers_in_archive":2}],"subtasks":[],"parent_tasks":[{"url":"/task/text-to-image","name":"Text-To-Image"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":28,"of":28,"tagged_in_all":42,"items":[{"url":"/paper/character-centric-story-visualization-via","title":"Character-Centric Story Visualization via Visual Planning and Token Alignment","date":"2022-10-16","arxiv_id":"2210.08465","repositories_listed":2,"syntology":{"n":21,"n_ran":12,"n_unverified":9,"n_pointer_only":0}},{"url":"/paper/character-preserving-coherent-story","title":"Character-Preserving Coherent Story Visualization","date":"2020-08-01","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/consistent-story-generation-with-asymmetry","title":"Consistent Story Generation with Asymmetry Zigzag Sampling","date":"2025-06-11","arxiv_id":"2506.09612","repositories_listed":1,"syntology":null},{"url":"/paper/vistorybench-comprehensive-benchmark-suite","title":"ViStoryBench: Comprehensive Benchmark Suite for Story Visualization","date":"2025-05-30","arxiv_id":"2505.24862","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/storyweaver-a-unified-world-model-for","title":"StoryWeaver: A Unified World Model for Knowledge-Enhanced Story Character Customization","date":"2024-12-10","arxiv_id":"2412.07375","repositories_listed":1,"syntology":null},{"url":"/paper/story-adapter-a-training-free-iterative","title":"Story-Adapter: A Training-free Iterative Framework for Long Story Visualization","date":"2024-10-08","arxiv_id":"2410.06244","repositories_listed":1,"syntology":{"n":7,"n_ran":2,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/temporalstory-enhancing-consistency-in-story","title":"ContextualStory: Consistent Visual Storytelling with Spatially-Enhanced and Storyline Context","date":"2024-07-13","arxiv_id":"2407.09774","repositories_listed":1,"syntology":null},{"url":"/paper/boosting-consistency-in-story-visualization","title":"Boosting Consistency in Story Visualization with Rich-Contextual Conditional Diffusion Models","date":"2024-07-02","arxiv_id":"2407.02482","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_unverified":1,"n_pointer_only":4}},{"url":"/paper/storyimager-a-unified-and-efficient-framework","title":"StoryImager: A Unified and Efficient Framework for Coherent Story Visualization and Completion","date":"2024-04-09","arxiv_id":"2404.05979","repositories_listed":1,"syntology":null},{"url":"/paper/masked-generative-story-transformer-with","title":"Masked Generative Story Transformer with Character Guidance and Caption Augmentation","date":"2024-03-13","arxiv_id":"2403.08502","repositories_listed":1,"syntology":null},{"url":"/paper/training-free-consistent-text-to-image","title":"Training-Free Consistent Text-to-Image Generation","date":"2024-02-05","arxiv_id":"2402.03286","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/large-language-models-as-consistent-story","title":"StoryGPT-V: Large Language Models as Consistent Story Visualizers","date":"2023-12-04","arxiv_id":"2312.02252","repositories_listed":1,"syntology":{"n":15,"n_ran":9,"n_unverified":6,"n_pointer_only":15}},{"url":"/paper/autostory-generating-diverse-storytelling","title":"AutoStory: Generating Diverse Storytelling Images with Minimal Human Effort","date":"2023-11-19","arxiv_id":"2311.11243","repositories_listed":1,"syntology":null},{"url":"/paper/the-chosen-one-consistent-characters-in-text","title":"The Chosen One: Consistent Characters in Text-to-Image Diffusion Models","date":"2023-11-16","arxiv_id":"2311.10093","repositories_listed":1,"syntology":null},{"url":"/paper/storybench-a-multifaceted-benchmark-for-1","title":"StoryBench: A Multifaceted Benchmark for Continuous Story Visualization","date":"2023-08-22","arxiv_id":"2308.11606","repositories_listed":1,"syntology":{"n":16,"n_ran":14,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/story-visualization-by-online-text","title":"Story Visualization by Online Text Augmentation with Context Memory","date":"2023-08-15","arxiv_id":"2308.07575","repositories_listed":1,"syntology":null},{"url":"/paper/intelligent-grimm-open-ended-visual","title":"Intelligent Grimm -- Open-ended Visual Storytelling via Latent Diffusion Models","date":"2023-06-01","arxiv_id":"2306.00973","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/talecrafter-interactive-story-visualization","title":"TaleCrafter: Interactive Story Visualization with Multiple Characters","date":"2023-05-29","arxiv_id":"2305.18247","repositories_listed":1,"syntology":null},{"url":"/paper/make-a-story-visual-memory-conditioned","title":"Make-A-Story: Visual Memory Conditioned Consistent Story Generation","date":"2022-11-23","arxiv_id":"2211.13319","repositories_listed":1,"syntology":{"n":8,"n_ran":0,"n_unverified":8,"n_pointer_only":0}},{"url":"/paper/synthesizing-coherent-story-with-auto","title":"Synthesizing Coherent Story with Auto-Regressive Latent Diffusion Models","date":"2022-11-20","arxiv_id":"2211.10950","repositories_listed":1,"syntology":null},{"url":"/paper/storydall-e-adapting-pretrained-text-to-image","title":"StoryDALL-E: Adapting Pretrained Text-to-Image Transformers for Story Continuation","date":"2022-09-13","arxiv_id":"2209.06192","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/word-level-fine-grained-story-visualization","title":"Word-Level Fine-Grained Story Visualization","date":"2022-08-03","arxiv_id":"2208.02341","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/modular-storygan-with-background-and-theme","title":"Modular StoryGAN with Background and Theme Awareness for Story Visualization","date":"2022-06-02","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/integrating-visuospatial-linguistic-and-1","title":"Integrating Visuospatial, Linguistic, and Commonsense Structure into Story Visualization","date":"2021-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/integrating-visuospatial-linguistic-and","title":"Integrating Visuospatial, Linguistic and Commonsense Structure into Story Visualization","date":"2021-10-21","arxiv_id":"2110.10834","repositories_listed":1,"syntology":null},{"url":"/paper/improving-generation-and-evaluation-of-visual","title":"Improving Generation and Evaluation of Visual Stories via Semantic Consistency","date":"2021-05-20","arxiv_id":"2105.10026","repositories_listed":1,"syntology":{"n":14,"n_ran":3,"n_unverified":11,"n_pointer_only":0}},{"url":"/paper/storygan-a-sequential-conditional-gan-for","title":"StoryGAN: A Sequential Conditional GAN for Story Visualization","date":"2018-12-06","arxiv_id":"1812.02784","repositories_listed":1,"syntology":null},{"url":"/paper/show-me-a-story-towards-coherent-neural-story","title":"Show Me a Story: Towards Coherent Neural Story Illustration","date":"2018-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null}],"syntology_records":12,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}