{"url":"/task/story-continuation","name":"Story Continuation","slug":"story-continuation","description_markdown":"The task involves providing an initial scene that can be obtained in real world use cases. By including this scene, a model can then copy and adapt elements from it as it generates subsequent images.\r\n\r\nSource: [StoryDALL-E: Adapting Pretrained Text-to-Image Transformers for Story Continuation](https://paperswithcode.com/paper/storydall-e-adapting-pretrained-text-to-image)","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":10,"papers_with_code":6,"benchmarks":3,"benchmark_tables_in_archive":3,"benchmark_tables_shown":3,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":1,"subtasks":0,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/story-continuation-on-flintstonessv","slug":"story-continuation-on-flintstonessv","dataset":"FlintstonesSV","dataset_url":null,"rows_in_archive":6,"metrics":["FID","Char-F1","F-Acc"],"first_row_in_archive_order":{"model":"ContextualStory","paper_title":"ContextualStory: Consistent Visual Storytelling with Spatially-Enhanced and Storyline Context","paper_url":"/paper/temporalstory-enhancing-consistency-in-story","paper_date":"2024-07-13","arxiv_id":"2407.09774","code_links":[{"title":"sixiaozheng/contextualstory","url":"https://github.com/sixiaozheng/contextualstory"}],"syntology":null}},{"leaderboard":"/sota/story-continuation-on-pororosv","slug":"story-continuation-on-pororosv","dataset":"PororoSV","dataset_url":null,"rows_in_archive":6,"metrics":["FID","Char-F1","F-Acc"],"first_row_in_archive_order":{"model":"ContextualStory","paper_title":"ContextualStory: Consistent Visual Storytelling with Spatially-Enhanced and Storyline Context","paper_url":"/paper/temporalstory-enhancing-consistency-in-story","paper_date":"2024-07-13","arxiv_id":"2407.09774","code_links":[{"title":"sixiaozheng/contextualstory","url":"https://github.com/sixiaozheng/contextualstory"}],"syntology":null}},{"leaderboard":"/sota/story-continuation-on-vist","slug":"story-continuation-on-vist","dataset":"VIST","dataset_url":"/dataset/vist","rows_in_archive":2,"metrics":["FID"],"first_row_in_archive_order":{"model":"AR-LDM (SIS captions)","paper_title":"Synthesizing Coherent Story with Auto-Regressive Latent Diffusion Models","paper_url":"/paper/synthesizing-coherent-story-with-auto","paper_date":"2022-11-20","arxiv_id":"2211.10950","code_links":[{"title":"xichenpan/ARLDM","url":"https://github.com/xichenpan/ARLDM"}],"syntology":null}}],"datasets":[{"url":"/dataset/vist","name":"VIST","full_name":"Visual Storytelling","num_papers_in_archive":107}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":6,"of":6,"tagged_in_all":10,"items":[{"url":"/paper/temporalstory-enhancing-consistency-in-story","title":"ContextualStory: Consistent Visual Storytelling with Spatially-Enhanced and Storyline Context","date":"2024-07-13","arxiv_id":"2407.09774","repositories_listed":1,"syntology":null},{"url":"/paper/how-far-can-we-extract-diverse-perspectives","title":"How Far Can We Extract Diverse Perspectives from Large Language Models?","date":"2023-11-16","arxiv_id":"2311.09799","repositories_listed":1,"syntology":null},{"url":"/paper/storybench-a-multifaceted-benchmark-for-1","title":"StoryBench: A Multifaceted Benchmark for Continuous Story Visualization","date":"2023-08-22","arxiv_id":"2308.11606","repositories_listed":1,"syntology":{"n":16,"n_ran":14,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/conveying-the-predicted-future-to-users-a","title":"Conveying the Predicted Future to Users: A Case Study of Story Plot Prediction","date":"2023-02-17","arxiv_id":"2302.09122","repositories_listed":1,"syntology":null},{"url":"/paper/synthesizing-coherent-story-with-auto","title":"Synthesizing Coherent Story with Auto-Regressive Latent Diffusion Models","date":"2022-11-20","arxiv_id":"2211.10950","repositories_listed":1,"syntology":null},{"url":"/paper/storydall-e-adapting-pretrained-text-to-image","title":"StoryDALL-E: Adapting Pretrained Text-to-Image Transformers for Story Continuation","date":"2022-09-13","arxiv_id":"2209.06192","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":0}}],"syntology_records":2,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":1,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}