{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/temporalstory-enhancing-consistency-in-story","title":"ContextualStory: Consistent Visual Storytelling with Spatially-Enhanced and Storyline Context","arxiv_id":"2407.09774","date":"2024-07-13","proceeding":null,"authors":["Sixiao Zheng","Yanwei Fu"],"abstract":"Visual storytelling involves generating a sequence of coherent frames from a textual storyline while maintaining consistency in characters and scenes. Existing autoregressive methods, which rely on previous frame-sentence pairs, struggle with high memory usage, slow generation speeds, and limited context integration. To address these issues, we propose ContextualStory, a novel framework designed to generate coherent story frames and extend frames for visual storytelling. ContextualStory utilizes Spatially-Enhanced Temporal Attention to capture spatial and temporal dependencies, handling significant character movements effectively. Additionally, we introduce a Storyline Contextualizer to enrich context in storyline embedding, and a StoryFlow Adapter to measure scene changes between frames for guiding the model. Extensive experiments on PororoSV and FlintstonesSV datasets demonstrate that ContextualStory significantly outperforms existing SOTA methods in both story visualization and continuation. Code is available at https://github.com/sixiaozheng/ContextualStory.","url_abs":"https://arxiv.org/abs/2407.09774v3","url_pdf":"https://arxiv.org/pdf/2407.09774v3.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"temporalstory-enhancing-consistency-in-story","repo_url":"https://github.com/sixiaozheng/contextualstory","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"pytorch","reach":null}],"tasks":[{"task_slug":"image-generation","task_name":"Image Generation"},{"task_slug":"story-continuation","task_name":"Story Continuation"},{"task_slug":"story-visualization","task_name":"Story Visualization"},{"task_slug":"text-to-image-generation","task_name":"Text-to-Image Generation"},{"task_slug":"visual-storytelling","task_name":"Visual Storytelling"}],"methods":[{"method_slug":"adapter","method_name":"Adapter"},{"method_slug":"attention","method_name":"Attention"},{"method_slug":"softmax","method_name":"Softmax"}],"datasets_introduced":[],"methods_introduced":[],"results":[{"leaderboard":"/sota/story-continuation-on-flintstonessv","task":"Story Continuation","dataset":"FlintstonesSV","model":"ContextualStory","rank_in_archive_order":1,"of":6,"metrics":{"FID":"16.33"},"uses_additional_data":false},{"leaderboard":"/sota/story-continuation-on-pororosv","task":"Story Continuation","dataset":"PororoSV","model":"ContextualStory","rank_in_archive_order":1,"of":6,"metrics":{"FID":"14.20"},"uses_additional_data":false},{"leaderboard":"/sota/story-visualization-on-pororo","task":"Story Visualization","dataset":"Pororo","model":"ContextualStory","rank_in_archive_order":1,"of":5,"metrics":{"FID":"14.07"},"uses_additional_data":false}],"syntology":{"syntology_url":"https://syntology.ai/paper/2407.09774","atlas_url":"https://app.syntology.ai/?focus=2407.09774","mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}