{"url":"/sota/video-generation-on-bair-robot-pushing","task":{"name":"Video Generation","url":"/task/video-generation","note":null},"dataset":{"name":"BAIR Robot Pushing","url":"/dataset/bair-robot-pushing"},"category":"Computer Vision","categories":["Computer Vision","Natural Language Processing"],"category_note":null,"description":"<span style=\"color:grey; opacity: 0.6\">( Various Video Generation Tasks.\r\nGif credit: [MaGViT](https://paperswithcode.com/paper/magvit-masked-generative-video-transformer) )</span>","description_from":"task","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","rank":"the archive's row order at snapshot; not re-ranked","rows_end_at":"2025-07-28","rows_withheld_as_spam":0,"metric_values":"the archive's strings, untouched"},"metrics":["FVD score","SSIM","PSNR","LPIPS","Cond","Train","Pred","Notes"],"metric_direction":{"note":"inferred from the metric name only (the archive records no direction); null = not inferred, chart draws points only","by_metric":{"FVD score":"higher","SSIM":"higher","PSNR":"higher","LPIPS":null,"Cond":null,"Train":null,"Pred":null,"Notes":null}},"counts":{"rows":31,"rows_with_code":30,"rows_with_paper_page":31,"rows_dated":31,"rows_using_additional_data":1},"rows":[{"rank_in_archive_order":1,"model":"MAGVIT","metrics":{"Cond":"1","FVD score":"62","Pred":"15","Train":"15"},"uses_additional_data":false,"paper_date":"2022-12-10","paper":"/paper/magvit-masked-generative-video-transformer","paper_url":"https://arxiv.org/abs/2212.05199v2","paper_title":"MAGVIT: Masked Generative Video Transformer","code":"https://github.com/google-research/magvit","n_code_links":1,"syntology":{"n_ran":1,"n_unverified":9,"n_samples":10,"n_pointer_only_licence":0}},{"rank_in_archive_order":2,"model":"RaMViD","metrics":{"Cond":"1","FVD score":"84.20","Pred":"15","Train":"20"},"uses_additional_data":false,"paper_date":"2022-06-15","paper":"/paper/diffusion-models-for-video-prediction-and","paper_url":"https://arxiv.org/abs/2206.07696v3","paper_title":"Diffusion Models for Video Prediction and Infilling","code":"https://github.com/Tobi-r9/RaMViD","n_code_links":1,"syntology":{"n_ran":9,"n_unverified":10,"n_samples":19,"n_pointer_only_licence":0}},{"rank_in_archive_order":3,"model":"NUWA","metrics":{"Cond":"1","FVD score":"86.9","Pred":"15","Train":"15"},"uses_additional_data":true,"paper_date":"2021-11-24","paper":"/paper/nuwa-visual-synthesis-pre-training-for-neural","paper_url":"https://arxiv.org/abs/2111.12417v1","paper_title":"NÜWA: Visual Synthesis Pre-training for Neural visUal World creAtion","code":"https://github.com/lucidrains/nuwa-pytorch","n_code_links":1,"syntology":{"n_ran":3,"n_unverified":0,"n_samples":3,"n_pointer_only_licence":2}},{"rank_in_archive_order":4,"model":"MCVD : c2t5p14","metrics":{"Cond":"2","FVD score":"87.9","PSNR":"19.1","Pred":"14","SSIM":"0.838","Train":"5"},"uses_additional_data":false,"paper_date":"2022-05-19","paper":"/paper/masked-conditional-video-diffusion-for","paper_url":"https://arxiv.org/abs/2205.09853v4","paper_title":"MCVD: Masked Conditional Video Diffusion for Prediction, Generation, and Interpolation","code":"https://github.com/voletiv/mcvd-pytorch","n_code_links":2,"syntology":{"n_ran":7,"n_unverified":9,"n_samples":16,"n_pointer_only_licence":0}},{"rank_in_archive_order":5,"model":"MCVD : c1t5p15","metrics":{"Cond":"1","FVD score":"89.5","PSNR":"16.9","Pred":"15","SSIM":"0.78","Train":"5"},"uses_additional_data":false,"paper_date":"2022-05-19","paper":"/paper/masked-conditional-video-diffusion-for","paper_url":"https://arxiv.org/abs/2205.09853v4","paper_title":"MCVD: Masked Conditional Video Diffusion for Prediction, Generation, and Interpolation","code":"https://github.com/voletiv/mcvd-pytorch","n_code_links":2,"syntology":{"n_ran":7,"n_unverified":9,"n_samples":16,"n_pointer_only_licence":0}},{"rank_in_archive_order":6,"model":"FitVid","metrics":{"Cond":"1","FVD score":"93.6","Notes":"Uses 100 times more fake than real samples (atypical)","Pred":"15","Train":"15"},"uses_additional_data":false,"paper_date":"2021-06-24","paper":"/paper/fitvid-overfitting-in-pixel-level-video","paper_url":"https://arxiv.org/abs/2106.13195v1","paper_title":"FitVid: Overfitting in Pixel-Level Video Prediction","code":"https://github.com/google-research/fitvid","n_code_links":1,"syntology":{"n_ran":0,"n_unverified":11,"n_samples":11,"n_pointer_only_licence":0}},{"rank_in_archive_order":7,"model":"Video Transformer","metrics":{"Cond":"1","FVD score":"94± 2","Notes":"FVD on only leftmost samples is 94, FVD on unrolled (all subsequences) is 96","Pred":"15","Train":"15"},"uses_additional_data":false,"paper_date":"2019-06-06","paper":"/paper/scaling-autoregressive-video-models","paper_url":"https://arxiv.org/abs/1906.02634v3","paper_title":"Scaling Autoregressive Video Models","code":"https://github.com/rakhimovv/lvt","n_code_links":1,"syntology":null},{"rank_in_archive_order":8,"model":"CCVS","metrics":{"Cond":"1","FVD score":"99 ± 2","Pred":"15","Train":"15"},"uses_additional_data":false,"paper_date":"2021-07-16","paper":"/paper/ccvs-context-aware-controllable-video","paper_url":"https://arxiv.org/abs/2107.08037v2","paper_title":"CCVS: Context-aware Controllable Video Synthesis","code":"https://github.com/16lemoing/ccvs","n_code_links":1,"syntology":{"n_ran":3,"n_unverified":5,"n_samples":8,"n_pointer_only_licence":0}},{"rank_in_archive_order":9,"model":"VideoGPT","metrics":{"Cond":"1","FVD score":"103.3","Pred":"15","Train":"15"},"uses_additional_data":false,"paper_date":"2021-04-20","paper":"/paper/videogpt-video-generation-using-vq-vae-and","paper_url":"https://arxiv.org/abs/2104.10157v2","paper_title":"VideoGPT: Video Generation using VQ-VAE and Transformers","code":"https://github.com/wilson1yan/VideoGPT","n_code_links":3,"syntology":null},{"rank_in_archive_order":10,"model":"TrIVD-GAN-FP","metrics":{"Cond":"1","FVD score":"103.3","Pred":"15","Train":"15"},"uses_additional_data":false,"paper_date":"2020-03-09","paper":"/paper/transformation-based-adversarial-video","paper_url":"https://arxiv.org/abs/2003.04035v3","paper_title":"Transformation-based Adversarial Video Prediction on Large-Scale Data","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":11,"model":"DVD-GAN-FP","metrics":{"Cond":"1","FVD score":"109.8","Pred":"15","Train":"15"},"uses_additional_data":false,"paper_date":"2019-07-15","paper":"/paper/efficient-video-generation-on-complex","paper_url":"https://arxiv.org/abs/1907.06571v2","paper_title":"Adversarial Video Generation on Complex Datasets","code":"https://github.com/Harrypotterrrr/DVD-GAN","n_code_links":1,"syntology":null},{"rank_in_archive_order":12,"model":"SAVP (from FVD)","metrics":{"Cond":"2","FVD score":"116.4","Pred":"14","Train":"14"},"uses_additional_data":false,"paper_date":"2018-04-04","paper":"/paper/stochastic-adversarial-video-prediction","paper_url":"http://arxiv.org/abs/1804.01523v1","paper_title":"Stochastic Adversarial Video Prediction","code":"https://github.com/alexlee-gk/video_prediction","n_code_links":4,"syntology":{"n_ran":4,"n_unverified":12,"n_samples":16,"n_pointer_only_licence":3}},{"rank_in_archive_order":13,"model":"MCVD : c2t5p28","metrics":{"Cond":"2","FVD score":"118.4","PSNR":"16.2","Pred":"28","SSIM":"0.745","Train":"5"},"uses_additional_data":false,"paper_date":"2022-05-19","paper":"/paper/masked-conditional-video-diffusion-for","paper_url":"https://arxiv.org/abs/2205.09853v4","paper_title":"MCVD: Masked Conditional Video Diffusion for Prediction, Generation, and Interpolation","code":"https://github.com/voletiv/mcvd-pytorch","n_code_links":2,"syntology":{"n_ran":7,"n_unverified":9,"n_samples":16,"n_pointer_only_licence":0}},{"rank_in_archive_order":14,"model":"LVT","metrics":{"Cond":"1","FVD score":"125.76±2.90","Pred":"15","Train":"15"},"uses_additional_data":false,"paper_date":"2020-06-18","paper":"/paper/latent-video-transformer","paper_url":"https://arxiv.org/abs/2006.10704v1","paper_title":"Latent Video Transformer","code":"https://github.com/rakhimovv/lvt","n_code_links":1,"syntology":{"n_ran":1,"n_unverified":4,"n_samples":5,"n_pointer_only_licence":0}},{"rank_in_archive_order":15,"model":"VideoFlow","metrics":{"Cond":"3","FVD score":"131±5","Pred":"14 (total 16)","Train":"10"},"uses_additional_data":false,"paper_date":"2019-03-04","paper":"/paper/videoflow-a-flow-based-generative-model-for","paper_url":"https://arxiv.org/abs/1903.01434v3","paper_title":"VideoFlow: A Conditional Flow-Based Model for Stochastic Video Generation","code":"https://github.com/tensorflow/tensor2tensor","n_code_links":1,"syntology":null},{"rank_in_archive_order":16,"model":"Hier-VRNN","metrics":{"Cond":"2","FVD score":"143.4","LPIPS":"0.055±0.03","Pred":"28","SSIM":"0.822±0.06","Train":"10"},"uses_additional_data":false,"paper_date":"2019-04-27","paper":"/paper/improved-conditional-vrnns-for-video","paper_url":"http://arxiv.org/abs/1904.12165v1","paper_title":"Improved Conditional VRNNs for Video Prediction","code":"https://github.com/facebookresearch/improved_vrnn","n_code_links":1,"syntology":null},{"rank_in_archive_order":17,"model":"SAVP (from vRNN)","metrics":{"Cond":"2","FVD score":"143.43","LPIPS":"0.062±0.03","Pred":"28","SSIM":"0.795±0.07","Train":"10"},"uses_additional_data":false,"paper_date":"2018-04-04","paper":"/paper/stochastic-adversarial-video-prediction","paper_url":"http://arxiv.org/abs/1804.01523v1","paper_title":"Stochastic Adversarial Video Prediction","code":"https://github.com/alexlee-gk/video_prediction","n_code_links":4,"syntology":{"n_ran":4,"n_unverified":12,"n_samples":16,"n_pointer_only_licence":3}},{"rank_in_archive_order":18,"model":"VRNN 1L","metrics":{"Cond":"2","FVD score":"149.22","LPIPS":"0.058±0.03","Pred":"28","SSIM":"0.829±0.06","Train":"10"},"uses_additional_data":false,"paper_date":"2019-04-27","paper":"/paper/improved-conditional-vrnns-for-video","paper_url":"http://arxiv.org/abs/1904.12165v1","paper_title":"Improved Conditional VRNNs for Video Prediction","code":"https://github.com/facebookresearch/improved_vrnn","n_code_links":1,"syntology":null},{"rank_in_archive_order":19,"model":"SAVP (from SRVP)","metrics":{"Cond":"2","FVD score":"152±9","LPIPS":"0.0634±0.0026","PSNR":"18.44±0.25","Pred":"28","SSIM":"0.7887±0.0092","Train":"12"},"uses_additional_data":false,"paper_date":"2018-04-04","paper":"/paper/stochastic-adversarial-video-prediction","paper_url":"http://arxiv.org/abs/1804.01523v1","paper_title":"Stochastic Adversarial Video Prediction","code":"https://github.com/alexlee-gk/video_prediction","n_code_links":4,"syntology":{"n_ran":4,"n_unverified":12,"n_samples":16,"n_pointer_only_licence":3}},{"rank_in_archive_order":20,"model":"WAM","metrics":{"Cond":"2","FVD score":"159.6","LPIPS":"0.0936","PSNR":"21.02","Pred":"28","SSIM":"0.844","Train":"14"},"uses_additional_data":false,"paper_date":"2020-02-23","paper":"/paper/exploring-spatial-temporal-multi-frequency","paper_url":"https://arxiv.org/abs/2002.09905v2","paper_title":"Exploring Spatial-Temporal Multi-Frequency Analysis for High-Fidelity and Temporal-Consistency Video Prediction","code":"https://github.com/Bei-Jin/STMFANet","n_code_links":1,"syntology":null},{"rank_in_archive_order":21,"model":"SRVP","metrics":{"Cond":"2","FVD score":"162 ± 4","LPIPS":"0.0574±0.0032","PSNR":"19.59±0.27","Pred":"28","SSIM":"0.8196±0.0084","Train":"12"},"uses_additional_data":false,"paper_date":"2020-02-21","paper":"/paper/stochastic-latent-residual-video-prediction-1","paper_url":"https://arxiv.org/abs/2002.09219v4","paper_title":"Stochastic Latent Residual Video Prediction","code":"https://github.com/edouardelasalles/srvp","n_code_links":1,"syntology":{"n_ran":2,"n_unverified":10,"n_samples":12,"n_pointer_only_licence":0}},{"rank_in_archive_order":22,"model":"SLAMP","metrics":{"Cond":"2","FVD score":"245 ± 5","LPIPS":"0.0596±0.0032","PSNR":"19.67±0.26","Pred":"28","SSIM":"0.8175±0.084","Train":"10"},"uses_additional_data":false,"paper_date":"2021-08-05","paper":"/paper/slamp-stochastic-latent-appearance-and-motion","paper_url":"https://arxiv.org/abs/2108.02760v1","paper_title":"SLAMP: Stochastic Latent Appearance and Motion Prediction","code":"https://github.com/kaanakan/slamp","n_code_links":1,"syntology":{"n_ran":2,"n_unverified":0,"n_samples":2,"n_pointer_only_licence":0}},{"rank_in_archive_order":23,"model":"SVG (from SRVP)","metrics":{"Cond":"2","FVD score":"255±4","LPIPS":"0.0609±0.0034","PSNR":"18.95±0.26","Pred":"28","SSIM":"0.8058±0.0088","Train":"12"},"uses_additional_data":false,"paper_date":"2018-02-21","paper":"/paper/stochastic-video-generation-with-a-learned","paper_url":"http://arxiv.org/abs/1802.07687v2","paper_title":"Stochastic Video Generation with a Learned Prior","code":"https://github.com/edenton/svg","n_code_links":3,"syntology":{"n_ran":1,"n_unverified":1,"n_samples":2,"n_pointer_only_licence":2}},{"rank_in_archive_order":24,"model":"SVG-LP (from vRNN)","metrics":{"Cond":"2","FVD score":"256.62","LPIPS":"0.061±0.03","Pred":"28","SSIM":"0.816±0.07","Train":"10"},"uses_additional_data":false,"paper_date":"2018-02-21","paper":"/paper/stochastic-video-generation-with-a-learned","paper_url":"http://arxiv.org/abs/1802.07687v2","paper_title":"Stochastic Video Generation with a Learned Prior","code":"https://github.com/edenton/svg","n_code_links":3,"syntology":{"n_ran":1,"n_unverified":1,"n_samples":2,"n_pointer_only_licence":2}},{"rank_in_archive_order":25,"model":"SV2P (from FVD)","metrics":{"Cond":"2","FVD score":"262.5","Pred":"14","Train":"14"},"uses_additional_data":false,"paper_date":"2017-10-30","paper":"/paper/stochastic-variational-video-prediction","paper_url":"http://arxiv.org/abs/1710.11252v2","paper_title":"Stochastic Variational Video Prediction","code":"https://github.com/StanfordVL/roboturk_real_dataset","n_code_links":3,"syntology":null},{"rank_in_archive_order":26,"model":"CDNA (from FVD)","metrics":{"Cond":"2","FVD score":"296.5","Pred":"14","Train":"14"},"uses_additional_data":false,"paper_date":"2016-05-23","paper":"/paper/unsupervised-learning-for-physical","paper_url":"http://arxiv.org/abs/1605.07157v4","paper_title":"Unsupervised Learning for Physical Interaction through Video Prediction","code":"https://github.com/tensorflow/models/tree/master/research/video_prediction","n_code_links":2,"syntology":{"n_ran":2,"n_unverified":0,"n_samples":2,"n_pointer_only_licence":2}},{"rank_in_archive_order":27,"model":"SVG-FP (from FVD)","metrics":{"Cond":"2","FVD score":"315.5","Pred":"14","Train":"14"},"uses_additional_data":false,"paper_date":"2018-02-21","paper":"/paper/stochastic-video-generation-with-a-learned","paper_url":"http://arxiv.org/abs/1802.07687v2","paper_title":"Stochastic Video Generation with a Learned Prior","code":"https://github.com/edenton/svg","n_code_links":3,"syntology":{"n_ran":1,"n_unverified":1,"n_samples":2,"n_pointer_only_licence":2}},{"rank_in_archive_order":28,"model":"Baseline (from LVT)","metrics":{"Cond":"1","FVD score":"320.9","Pred":"15","Train":"15"},"uses_additional_data":false,"paper_date":"2020-06-18","paper":"/paper/latent-video-transformer","paper_url":"https://arxiv.org/abs/2006.10704v1","paper_title":"Latent Video Transformer","code":"https://github.com/rakhimovv/lvt","n_code_links":1,"syntology":{"n_ran":1,"n_unverified":4,"n_samples":5,"n_pointer_only_licence":0}},{"rank_in_archive_order":29,"model":"MoCoGAN","metrics":{"Cond":"4","FVD score":"503","Pred":"12","Train":"12"},"uses_additional_data":false,"paper_date":"2017-07-17","paper":"/paper/mocogan-decomposing-motion-and-content-for","paper_url":"http://arxiv.org/abs/1707.04993v2","paper_title":"MoCoGAN: Decomposing Motion and Content for Video Generation","code":"https://github.com/sergeytulyakov/mocogan","n_code_links":5,"syntology":null},{"rank_in_archive_order":30,"model":"SV2P (from SRVP)","metrics":{"Cond":"2","FVD score":"965±17","LPIPS":"0.0912±0.0053","PSNR":"20.39±0.27","Pred":"28","SSIM":"0.8169±0.0086","Train":"12"},"uses_additional_data":false,"paper_date":"2017-10-30","paper":"/paper/stochastic-variational-video-prediction","paper_url":"http://arxiv.org/abs/1710.11252v2","paper_title":"Stochastic Variational Video Prediction","code":"https://github.com/StanfordVL/roboturk_real_dataset","n_code_links":3,"syntology":null},{"rank_in_archive_order":31,"model":"SAVP-VAE (from WAM)","metrics":{"Cond":"2","PSNR":"19.09","Pred":"28","SSIM":"0.815","Train":"14"},"uses_additional_data":false,"paper_date":"2018-04-04","paper":"/paper/stochastic-adversarial-video-prediction","paper_url":"http://arxiv.org/abs/1804.01523v1","paper_title":"Stochastic Adversarial Video Prediction","code":"https://github.com/alexlee-gk/video_prediction","n_code_links":4,"syntology":{"n_ran":4,"n_unverified":12,"n_samples":16,"n_pointer_only_licence":3}}],"since_archive":{"claim":"Results that newer papers report for their own method, placed here by Syntology. A model pointed at the cell in the paper's own table; the number was read from that cell and checked against this leaderboard's metric, dataset, split and scale; an independent check that saw this leaderboard's other rows and every other leaderboard on the same dataset accepted it. Not reviewed by the paper's authors or by the archive's editors, and not ranked against the archive rows.","extraction_file_present":true,"measurement":{"test_papers":883,"papers_with_output":881,"judged_true":108,"judged":110,"wilson95_lower":0.9361,"measured_on":"2026-09-24","frozen_commit":"0e3de0df94"},"measurement_note":"blind adjudication of accepted entries on a held-out split of archive papers, rules frozen before the test","coverage":{"sentence":"Syntology has checked 6,264 of the 9,581 papers on this site that are newer than the archive; results from the others appear after they are checked.","complete":false,"papers_newer_than_archive":9581,"papers_checked":6264,"papers_extracted_not_yet_verified":0,"boards_without_verdict":2,"papers_not_yet_extracted":3316},"order":"newest first by month (arXiv date, else the arXiv-id month), then arXiv id descending","columns":[],"entries":[]},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per row: N of M harvested code samples from that row's paper executed on a synthesized fixture; the other M-N are unverified. Not a reproduction of the row's number; not a correctness claim. n_pointer_only_licence counts samples the site points at rather than redistributes (a licence axis, independent of ran/unverified).","rows_with_graph_line":20,"rows_with_any_sample_ran":19,"distinct_papers_with_graph_line":12,"distinct_papers_with_any_sample_ran":11,"samples_over_distinct_papers":{"n_ran":35,"n_unverified":71,"n_samples":106,"n_pointer_only_licence":9,"note":"each paper (arXiv id) counted once, however many rows it is behind; this is the page-level figure"},"samples_row_weighted":{"n_ran":64,"n_unverified":131,"n_samples":195,"n_pointer_only_licence":22,"note":"row-weighted: a paper behind several rows is counted once per row; inflated relative to samples_over_distinct_papers by design, kept for readers summing the per-row syntology blocks"}}}