{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/generating-videos-with-scene-dynamics","title":"Generating Videos with Scene Dynamics","arxiv_id":"1609.02612","date":"2016-09-08","proceeding":"NeurIPS 2016 12","authors":["Carl Vondrick","Hamed Pirsiavash","Antonio Torralba"],"abstract":"We capitalize on large amounts of unlabeled video in order to learn a model\nof scene dynamics for both video recognition tasks (e.g. action classification)\nand video generation tasks (e.g. future prediction). We propose a generative\nadversarial network for video with a spatio-temporal convolutional architecture\nthat untangles the scene's foreground from the background. Experiments suggest\nthis model can generate tiny videos up to a second at full frame rate better\nthan simple baselines, and we show its utility at predicting plausible futures\nof static images. Moreover, experiments and visualizations show the model\ninternally learns useful features for recognizing actions with minimal\nsupervision, suggesting scene dynamics are a promising signal for\nrepresentation learning. We believe generative video models can impact many\napplications in video understanding and simulation.","url_abs":"http://arxiv.org/abs/1609.02612v3","url_pdf":"http://arxiv.org/pdf/1609.02612v3.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[],"tasks":[{"task_slug":"action-classification","task_name":"Action Classification"},{"task_slug":"future-prediction","task_name":"Future prediction"},{"task_slug":"classification","task_name":"General Classification"},{"task_slug":null,"task_name":"Generative Adversarial Network"},{"task_slug":"representation-learning","task_name":"Representation Learning"},{"task_slug":"self-supervised-action-recognition","task_name":"Self-Supervised Action Recognition"},{"task_slug":"video-generation","task_name":"Video Generation"},{"task_slug":"video-recognition","task_name":"Video Recognition"},{"task_slug":"video-understanding","task_name":"Video Understanding"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[{"leaderboard":"/sota/self-supervised-action-recognition-on-ucf101","task":"Self-Supervised Action Recognition","dataset":"UCF101","model":"VideoGan (C3D)","rank_in_archive_order":51,"of":53,"metrics":{"3-fold Accuracy":"52.1","Frozen":"false","Pre-Training Dataset":"UCF101"},"uses_additional_data":false},{"leaderboard":"/sota/video-generation-on-ucf-101-16-frames-64x64","task":"Video Generation","dataset":"UCF-101 16 frames, 64x64, Unconditional","model":"VGAN","rank_in_archive_order":7,"of":7,"metrics":{"Inception Score":"8.18"},"uses_additional_data":false},{"leaderboard":"/sota/video-generation-on-ucf-101-16-frames","task":"Video Generation","dataset":"UCF-101 16 frames, Unconditional, Single GPU","model":"VGAN","rank_in_archive_order":7,"of":7,"metrics":{"Inception Score":"8.18"},"uses_additional_data":false}],"syntology":{"syntology_url":null,"atlas_url":"https://app.syntology.ai/?focus=1609.02612","mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}