{"url":"/sota/video-prediction-on-kth","task":{"name":"Video Prediction","url":"/task/video-prediction","note":null},"dataset":{"name":"KTH","url":"/dataset/kth"},"category":"Computer Vision","categories":["Computer Vision","Time Series"],"category_note":null,"description":null,"description_from":null,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","rank":"the archive's row order at snapshot; not re-ranked","rows_end_at":"2025-07-28","rows_withheld_as_spam":0,"metric_values":"the archive's strings, untouched"},"metrics":["FVD","SSIM","PSNR","LPIPS","Cond","Train","Pred","Params (M)","MSE","Diversity"],"metric_direction":{"note":"inferred from the metric name only (the archive records no direction); null = not inferred, chart draws points only","by_metric":{"FVD":null,"SSIM":"higher","PSNR":"higher","LPIPS":null,"Cond":null,"Train":null,"Pred":null,"Params (M)":"lower","MSE":"lower","Diversity":null}},"counts":{"rows":31,"rows_with_code":29,"rows_with_paper_page":31,"rows_dated":31,"rows_using_additional_data":0},"rows":[{"rank_in_archive_order":1,"model":"Grid-keypoints","metrics":{"Cond":"10","FVD":"144.2","LPIPS":"0.092","PSNR":"27.11","Params (M)":"2.0","Pred":"40","SSIM":"0.837","Train":"10"},"uses_additional_data":false,"paper_date":"2021-07-28","paper":"/paper/accurate-grid-keypoint-learning-for-efficient","paper_url":"https://arxiv.org/abs/2107.13170v1","paper_title":"Accurate Grid Keypoint Learning for Efficient Video Prediction","code":"https://github.com/xjgaocs/Grid-Keypoint-Learning","n_code_links":1,"syntology":{"n_ran":2,"n_unverified":0,"n_samples":2,"n_pointer_only_licence":2}},{"rank_in_archive_order":2,"model":"SAVP-VAE (from Grid-keypoints)","metrics":{"Cond":"10","FVD":"145.7","LPIPS":"0.116","PSNR":"26.00","Params (M)":"7.3","Pred":"40","SSIM":"0.806","Train":"10"},"uses_additional_data":false,"paper_date":"2018-04-04","paper":"/paper/stochastic-adversarial-video-prediction","paper_url":"http://arxiv.org/abs/1804.01523v1","paper_title":"Stochastic Adversarial Video Prediction","code":"https://github.com/alexlee-gk/video_prediction","n_code_links":4,"syntology":{"n_ran":4,"n_unverified":12,"n_samples":16,"n_pointer_only_licence":3}},{"rank_in_archive_order":3,"model":"SVG-LP (from Grid-keypoints)","metrics":{"Cond":"10","FVD":"157.9","LPIPS":"0.129","PSNR":"23.91","Params (M)":"22.8","Pred":"40","SSIM":"0.800","Train":"10"},"uses_additional_data":false,"paper_date":"2018-02-21","paper":"/paper/stochastic-video-generation-with-a-learned","paper_url":"http://arxiv.org/abs/1802.07687v2","paper_title":"Stochastic Video Generation with a Learned Prior","code":"https://github.com/edenton/svg","n_code_links":3,"syntology":{"n_ran":1,"n_unverified":1,"n_samples":2,"n_pointer_only_licence":2}},{"rank_in_archive_order":4,"model":"SAVP (from Grid-keypoints)","metrics":{"Cond":"10","FVD":"183.7","LPIPS":"0.126","PSNR":"23.79","Params (M)":"17.6","Pred":"40","SSIM":"0.699","Train":"10"},"uses_additional_data":false,"paper_date":"2018-04-04","paper":"/paper/stochastic-adversarial-video-prediction","paper_url":"http://arxiv.org/abs/1804.01523v1","paper_title":"Stochastic Adversarial Video Prediction","code":"https://github.com/alexlee-gk/video_prediction","n_code_links":4,"syntology":{"n_ran":4,"n_unverified":12,"n_samples":16,"n_pointer_only_licence":3}},{"rank_in_archive_order":5,"model":"SV2P time-invariant (from Grid-keypoints)","metrics":{"Cond":"10","FVD":"209.5","LPIPS":"0.232","PSNR":"25.87","Params (M)":"8.3","Pred":"40","SSIM":"0.782","Train":"10"},"uses_additional_data":false,"paper_date":"2017-10-30","paper":"/paper/stochastic-variational-video-prediction","paper_url":"http://arxiv.org/abs/1710.11252v2","paper_title":"Stochastic Variational Video Prediction","code":"https://github.com/StanfordVL/roboturk_real_dataset","n_code_links":3,"syntology":null},{"rank_in_archive_order":6,"model":"SRVP","metrics":{"Cond":"10","FVD":"222 ± 3","LPIPS":"0.0736±0.0029","PSNR":"29.69±032","Pred":"30","SSIM":"0.8697±0.0046","Train":"10"},"uses_additional_data":false,"paper_date":"2020-02-21","paper":"/paper/stochastic-latent-residual-video-prediction-1","paper_url":"https://arxiv.org/abs/2002.09219v4","paper_title":"Stochastic Latent Residual Video Prediction","code":"https://github.com/edouardelasalles/srvp","n_code_links":1,"syntology":{"n_ran":2,"n_unverified":10,"n_samples":12,"n_pointer_only_licence":0}},{"rank_in_archive_order":7,"model":"SLAMP","metrics":{"Cond":"10","FVD":"228 ± 5","LPIPS":"0.0795±0.0034","PSNR":"29.39±0.30","Pred":"30","SSIM":"0.8646±0.0050","Train":"10"},"uses_additional_data":false,"paper_date":"2021-08-05","paper":"/paper/slamp-stochastic-latent-appearance-and-motion","paper_url":"https://arxiv.org/abs/2108.02760v1","paper_title":"SLAMP: Stochastic Latent Appearance and Motion Prediction","code":"https://github.com/kaanakan/slamp","n_code_links":1,"syntology":{"n_ran":2,"n_unverified":0,"n_samples":2,"n_pointer_only_licence":0}},{"rank_in_archive_order":8,"model":"SV2P time-invariant (from Grid-keypoints)","metrics":{"Cond":"10","FVD":"253.5","LPIPS":"0.260","PSNR":"25.70","Params (M)":"8.3","Pred":"40","SSIM":"0.772","Train":"10"},"uses_additional_data":false,"paper_date":"2017-10-30","paper":"/paper/stochastic-variational-video-prediction","paper_url":"http://arxiv.org/abs/1710.11252v2","paper_title":"Stochastic Variational Video Prediction","code":"https://github.com/StanfordVL/roboturk_real_dataset","n_code_links":3,"syntology":null},{"rank_in_archive_order":9,"model":"SAVP (from SRVP)","metrics":{"Cond":"10","FVD":"374 ± 3","LPIPS":"0.1120±0.0039","PSNR":"26.51±0.29","Pred":"30","SSIM":"0.7564±0.0062","Train":"10"},"uses_additional_data":false,"paper_date":"2018-04-04","paper":"/paper/stochastic-adversarial-video-prediction","paper_url":"http://arxiv.org/abs/1804.01523v1","paper_title":"Stochastic Adversarial Video Prediction","code":"https://github.com/alexlee-gk/video_prediction","n_code_links":4,"syntology":{"n_ran":4,"n_unverified":12,"n_samples":16,"n_pointer_only_licence":3}},{"rank_in_archive_order":10,"model":"SVG-LP (from SRVP)","metrics":{"Cond":"10","FVD":"377 ± 6","LPIPS":"0.0923±0.0038","PSNR":"28.06±0.29","Pred":"30","SSIM":"0.8438±0.0054","Train":"10"},"uses_additional_data":false,"paper_date":"2018-02-21","paper":"/paper/stochastic-video-generation-with-a-learned","paper_url":"http://arxiv.org/abs/1802.07687v2","paper_title":"Stochastic Video Generation with a Learned Prior","code":"https://github.com/edenton/svg","n_code_links":3,"syntology":{"n_ran":1,"n_unverified":1,"n_samples":2,"n_pointer_only_licence":2}},{"rank_in_archive_order":11,"model":"Struct-VRNN (from Grid-keypoints)","metrics":{"Cond":"10","FVD":"395.0","LPIPS":"0.124","PSNR":"24.29","Params (M)":"2.3","Pred":"40","SSIM":"0.766","Train":"10"},"uses_additional_data":false,"paper_date":"2019-06-19","paper":"/paper/unsupervised-learning-of-object-structure-and","paper_url":"https://arxiv.org/abs/1906.07889v3","paper_title":"Unsupervised Learning of Object Structure and Dynamics from Videos","code":"https://github.com/google-research/google-research","n_code_links":1,"syntology":null},{"rank_in_archive_order":12,"model":"SV2P (from SRVP)","metrics":{"Cond":"10","FVD":"636 ± 1","LPIPS":"0.2049±0.0053","PSNR":"28.19±0.31","Pred":"30","SSIM":"0.838","Train":"10"},"uses_additional_data":false,"paper_date":"2017-10-30","paper":"/paper/stochastic-variational-video-prediction","paper_url":"http://arxiv.org/abs/1710.11252v2","paper_title":"Stochastic Variational Video Prediction","code":"https://github.com/StanfordVL/roboturk_real_dataset","n_code_links":3,"syntology":null},{"rank_in_archive_order":13,"model":"MSPred","metrics":{"LPIPS":"0.029","MSE":"23.18","PSNR":"27.81","SSIM":"0.951"},"uses_additional_data":false,"paper_date":"2022-03-17","paper":"/paper/video-prediction-at-multiple-scales-with","paper_url":"https://arxiv.org/abs/2203.09303v4","paper_title":"MSPred: Video Prediction at Multiple Spatio-Temporal Scales with Hierarchical Recurrent Networks","code":"https://github.com/AIS-Bonn/MSPred","n_code_links":1,"syntology":null},{"rank_in_archive_order":14,"model":"WAM","metrics":{"Cond":"10","PSNR":"29.85","Pred":"20","SSIM":"0.893"},"uses_additional_data":false,"paper_date":"2020-02-23","paper":"/paper/exploring-spatial-temporal-multi-frequency","paper_url":"https://arxiv.org/abs/2002.09905v2","paper_title":"Exploring Spatial-Temporal Multi-Frequency Analysis for High-Fidelity and Temporal-Consistency Video Prediction","code":"https://github.com/Bei-Jin/STMFANet","n_code_links":1,"syntology":null},{"rank_in_archive_order":15,"model":"E3d-LSTM","metrics":{"Cond":"10","PSNR":"29.31","Pred":"20","SSIM":"0.879"},"uses_additional_data":false,"paper_date":"2019-05-01","paper":"/paper/eidetic-3d-lstm-a-model-for-video-prediction","paper_url":"https://openreview.net/forum?id=B1lKS2AqtX","paper_title":"Eidetic 3D LSTM: A Model for Video Prediction and Beyond","code":"https://github.com/chengtan9907/simvpv2","n_code_links":3,"syntology":null},{"rank_in_archive_order":16,"model":"LMC","metrics":{"Cond":"10","LPIPS":"159.8","PSNR":"27.5","Pred":"40","SSIM":"0.879"},"uses_additional_data":false,"paper_date":"2021-04-02","paper":"/paper/video-prediction-recalling-long-term-motion","paper_url":"https://arxiv.org/abs/2104.00924v1","paper_title":"Video Prediction Recalling Long-term Motion Context via Memory Alignment Learning","code":"https://github.com/sangmin-git/LMC-Memory","n_code_links":1,"syntology":{"n_ran":0,"n_unverified":1,"n_samples":1,"n_pointer_only_licence":1}},{"rank_in_archive_order":17,"model":"MSNET","metrics":{"Cond":"10","PSNR":"27.08","Pred":"20","SSIM":"0.876"},"uses_additional_data":false,"paper_date":"2018-04-13","paper":"/paper/msnet-mutual-suppression-network-for","paper_url":"https://arxiv.org/abs/1804.04810v2","paper_title":"Mutual Suppression Network for Video Prediction using Disentangled Features","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":18,"model":"PredRNN++","metrics":{"Cond":"10","PSNR":"28.47","Pred":"20","SSIM":"0.865"},"uses_additional_data":false,"paper_date":"2018-04-17","paper":"/paper/predrnn-towards-a-resolution-of-the-deep-in","paper_url":"http://arxiv.org/abs/1804.06300v2","paper_title":"PredRNN++: Towards A Resolution of the Deep-in-Time Dilemma in Spatiotemporal Predictive Learning","code":"https://github.com/thuml/predrnn-pytorch","n_code_links":11,"syntology":{"n_ran":0,"n_unverified":7,"n_samples":7,"n_pointer_only_licence":0}},{"rank_in_archive_order":19,"model":"SAVP-VAE","metrics":{"Cond":"10","PSNR":"27.77","Pred":"20","SSIM":"0.852"},"uses_additional_data":false,"paper_date":"2018-04-04","paper":"/paper/stochastic-adversarial-video-prediction","paper_url":"http://arxiv.org/abs/1804.01523v1","paper_title":"Stochastic Adversarial Video Prediction","code":"https://github.com/alexlee-gk/video_prediction","n_code_links":4,"syntology":{"n_ran":4,"n_unverified":12,"n_samples":16,"n_pointer_only_licence":3}},{"rank_in_archive_order":20,"model":"VarNet","metrics":{"Cond":"10","PSNR":"28.48","Pred":"20","SSIM":"0.843"},"uses_additional_data":false,"paper_date":"2018-10-01","paper":"/paper/varnet-exploring-variations-for-unsupervised","paper_url":"https://ieeexplore.ieee.org/document/8594264","paper_title":"VarNet: Exploring Variations for Unsupervised Video Prediction","code":"https://github.com/jinbeibei/VarNet","n_code_links":1,"syntology":null},{"rank_in_archive_order":21,"model":"PredRNN-V2","metrics":{"Cond":"10","LPIPS":"0.139","PSNR":"28.37","Pred":"20","SSIM":"0.839"},"uses_additional_data":false,"paper_date":"2021-03-17","paper":"/paper/predrnn-a-recurrent-neural-network-for","paper_url":"https://arxiv.org/abs/2103.09504v4","paper_title":"PredRNN: A Recurrent Neural Network for Spatiotemporal Predictive Learning","code":"https://github.com/chengtan9907/simvpv2","n_code_links":3,"syntology":null},{"rank_in_archive_order":22,"model":"Znet","metrics":{"Cond":"10","PSNR":"27.58","Pred":"20","SSIM":"0.817"},"uses_additional_data":false,"paper_date":"2019-07-08","paper":"/paper/z-order-recurrent-neural-networks-for-video","paper_url":"https://ieeexplore.ieee.org/document/8784821","paper_title":"Z-Order Recurrent Neural Networks for Video Prediction","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":23,"model":"Conv-TT-LSTM","metrics":{"Cond":"10","LPIPS":"0.196","PSNR":"27.62","Pred":"20","SSIM":"0.815"},"uses_additional_data":false,"paper_date":"2020-02-21","paper":"/paper/convolutional-tensor-train-lstm-for-spatio","paper_url":"https://arxiv.org/abs/2002.09131v5","paper_title":"Convolutional Tensor-Train LSTM for Spatio-temporal Learning","code":"https://github.com/NVlabs/conv-tt-lstm","n_code_links":2,"syntology":{"n_ran":2,"n_unverified":0,"n_samples":2,"n_pointer_only_licence":2}},{"rank_in_archive_order":24,"model":"MCnet + Residual","metrics":{"Cond":"10","PSNR":"26.29","Pred":"20","SSIM":"0.806"},"uses_additional_data":false,"paper_date":"2017-06-25","paper":"/paper/decomposing-motion-and-content-for-natural","paper_url":"http://arxiv.org/abs/1706.08033v2","paper_title":"Decomposing Motion and Content for Natural Video Sequence Prediction","code":"https://github.com/rubenvillegas/iclr2017mcnet","n_code_links":1,"syntology":null},{"rank_in_archive_order":25,"model":"MCnet","metrics":{"Cond":"10","PSNR":"25.95","Pred":"20","SSIM":"0.804"},"uses_additional_data":false,"paper_date":"2017-06-25","paper":"/paper/decomposing-motion-and-content-for-natural","paper_url":"http://arxiv.org/abs/1706.08033v2","paper_title":"Decomposing Motion and Content for Natural Video Sequence Prediction","code":"https://github.com/rubenvillegas/iclr2017mcnet","n_code_links":1,"syntology":null},{"rank_in_archive_order":26,"model":"DFN","metrics":{"Cond":"10","PSNR":"27.26","Pred":"20","SSIM":"0.794"},"uses_additional_data":false,"paper_date":"2016-05-31","paper":"/paper/dynamic-filter-networks","paper_url":"http://arxiv.org/abs/1605.09673v2","paper_title":"Dynamic Filter Networks","code":"https://github.com/dbbert/dfn","n_code_links":1,"syntology":null},{"rank_in_archive_order":27,"model":"TrajGRU","metrics":{"Cond":"10","PSNR":"26.97","Pred":"20","SSIM":"0.790"},"uses_additional_data":false,"paper_date":"2017-06-12","paper":"/paper/deep-learning-for-precipitation-nowcasting-a","paper_url":"http://arxiv.org/abs/1706.03458v2","paper_title":"Deep Learning for Precipitation Nowcasting: A Benchmark and A New Model","code":"https://github.com/Hzzone/Precipitation-Nowcasting","n_code_links":4,"syntology":{"n_ran":9,"n_unverified":1,"n_samples":10,"n_pointer_only_licence":5}},{"rank_in_archive_order":28,"model":"fRNN","metrics":{"Cond":"10","PSNR":"26.12","Pred":"20","SSIM":"0.771"},"uses_additional_data":false,"paper_date":"2017-12-01","paper":"/paper/folded-recurrent-neural-networks-for-future","paper_url":"http://arxiv.org/abs/1712.00311v2","paper_title":"Folded Recurrent Neural Networks for Future Video Prediction","code":"https://github.com/moliusimon/frnn","n_code_links":1,"syntology":null},{"rank_in_archive_order":29,"model":"VPN","metrics":{"Cond":"10","PSNR":"23.76","Pred":"20","SSIM":"0.746"},"uses_additional_data":false,"paper_date":"2016-10-03","paper":"/paper/video-pixel-networks","paper_url":"http://arxiv.org/abs/1610.00527v1","paper_title":"Video Pixel Networks","code":"https://github.com/3ammor/Video-Pixel-Networks","n_code_links":1,"syntology":null},{"rank_in_archive_order":30,"model":"ConvLSTM","metrics":{"Cond":"10","LPIPS":"0.231","PSNR":"23.58","Pred":"20","SSIM":"0.712"},"uses_additional_data":false,"paper_date":"2015-06-13","paper":"/paper/convolutional-lstm-network-a-machine-learning","paper_url":"http://arxiv.org/abs/1506.04214v2","paper_title":"Convolutional LSTM Network: A Machine Learning Approach for Precipitation Nowcasting","code":"https://github.com/ndrplz/ConvLSTM_pytorch","n_code_links":23,"syntology":{"n_ran":1,"n_unverified":2,"n_samples":3,"n_pointer_only_licence":2}},{"rank_in_archive_order":31,"model":"DVG","metrics":{"Diversity":"0.483"},"uses_additional_data":false,"paper_date":"2021-07-09","paper":"/paper/diverse-video-generation-using-a-gaussian-1","paper_url":"https://arxiv.org/abs/2107.04619v1","paper_title":"Diverse Video Generation using a Gaussian Process Trigger","code":"https://github.com/shgaurav1/DVG","n_code_links":1,"syntology":null}],"since_archive":{"present":false,"note":"No Syntology-extracted rows are published in this build."},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per row: N of M harvested code samples from that row's paper executed on a synthesized fixture; the other M-N are unverified. Not a reproduction of the row's number; not a correctness claim. n_pointer_only_licence counts samples the site points at rather than redistributes (a licence axis, independent of ran/unverified).","rows_with_graph_line":14,"rows_with_any_sample_ran":12,"distinct_papers_with_graph_line":10,"distinct_papers_with_any_sample_ran":8,"samples_over_distinct_papers":{"n_ran":23,"n_unverified":34,"n_samples":57,"n_pointer_only_licence":17,"note":"each paper (arXiv id) counted once, however many rows it is behind; this is the page-level figure"},"samples_row_weighted":{"n_ran":36,"n_unverified":71,"n_samples":107,"n_pointer_only_licence":28,"note":"row-weighted: a paper behind several rows is counted once per row; inflated relative to samples_over_distinct_papers by design, kept for readers summing the per-row syntology blocks"}}}