{"url":"/dataset/moving-mnist","name":"Moving MNIST","full_name":null,"description_markdown":"The **Moving MNIST** dataset contains 10,000 video sequences, each consisting of 20 frames. In each video sequence, two digits move independently around the frame, which has a spatial resolution of 64×64 pixels. The digits frequently intersect with each other and bounce off the edges of the frame\r\n\r\nSource: [Mutual Suppression Network for Video Prediction using Disentangled Features](https://arxiv.org/abs/1804.04810)\r\nImage Source: [http://www.cs.toronto.edu/~nitish/unsupervised_video/](http://www.cs.toronto.edu/~nitish/unsupervised_video/)","description_withheld":null,"homepage":"http://www.cs.toronto.edu/~nitish/unsupervised_video/","introduced_date":"2015-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/unsupervised-learning-of-video","title":"Unsupervised Learning of Video Representations using LSTMs","first_author":"Nitish Srivastava","url":null},"license":{"name":"Unknown","url":null},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Videos","url":"/datasets/modality/videos"}],"tasks":[{"name":"Video Prediction","url":"/task/video-prediction","datasets_with_task":"/datasets/task/video-prediction"}],"languages":[],"variants":["Moving MNIST"],"data_loaders":[{"repo":"https://github.com/tensorflow/datasets","url":"https://www.tensorflow.org/datasets/catalog/moving_mnist","frameworks":["tf","jax"]}],"num_papers_in_archive":194,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/video-prediction-on-moving-mnist","task":"Video Prediction","dataset_variant":"Moving MNIST","rows":31,"metrics":["MSE","MAE","SSIM","LPIPS","PSNR"],"first_row_in_archive_order":{"model":"PredFormer","paper":"/paper/predformer-transformers-are-effective-spatial","metrics":{"MAE":"41.96","MSE":"11.62","PSNR":"39.89","SSIM":"0.9742"},"code_links":[{"title":"yyyujintang/predformer","url":"https://github.com/yyyujintang/predformer"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/gmg-a-video-prediction-method-based-on-global","title":"GMG: A Video Prediction Method Based on Global Focus and Motion Guided","date":"2025-03-14","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/predformer-transformers-are-effective-spatial","title":"Video Prediction Transformers without Recurrence or Convolution","date":"2024-10-07","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/implicit-stacked-autoregressive-model-for-1","title":"Implicit Stacked Autoregressive Model for Video Prediction","date":"2023-03-14","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/swinlstm-improving-spatiotemporal-prediction-1","title":"SwinLSTM: Improving Spatiotemporal Prediction Accuracy using Swin Transformer and LSTM","date":"2023-01-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/simvp-towards-simple-yet-powerful","title":"SimVPv2: Towards Simple yet Powerful Spatiotemporal Predictive Learning","date":"2022-11-22","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/efficient-multi-order-gated-aggregation","title":"MogaNet: Multi-order Gated Aggregation Network","date":"2022-11-07","rows_on_this_dataset":10,"code_links":7,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":12,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/temporal-attention-unit-towards-efficient","title":"Temporal Attention Unit: Towards Efficient Spatiotemporal Predictive Learning","date":"2022-06-24","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/simvp-simpler-yet-better-video-prediction-1","title":"SimVP: Simpler yet Better Video Prediction","date":"2022-06-09","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":6,"samples_unverified":3,"pointer_only_for_licence":9,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/video-prediction-at-multiple-scales-with","title":"MSPred: Video Prediction at Multiple Spatio-Temporal Scales with Hierarchical Recurrent Networks","date":"2022-03-17","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/mau-a-motion-aware-unit-for-video-prediction","title":"MAU: A Motion-Aware Unit for Video Prediction and Beyond","date":"2021-12-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/video-prediction-recalling-long-term-motion","title":"Video Prediction Recalling Long-term Motion Context via Memory Alignment Learning","date":"2021-04-02","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/predrnn-a-recurrent-neural-network-for","title":"PredRNN: A Recurrent Neural Network for Spatiotemporal Predictive Learning","date":"2021-03-17","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/efficient-and-information-preserving-future","title":"Efficient and Information-Preserving Future Frame Prediction and Beyond","date":"2020-05-01","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/self-attention-convlstm-for-spatiotemporal","title":"Self-Attention ConvLSTM for Spatiotemporal Prediction","date":"2020-04-03","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/disentangling-physical-dynamics-from-unknown","title":"Disentangling Physical Dynamics from Unknown Factors for Unsupervised Video Prediction","date":"2020-03-03","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/eidetic-3d-lstm-a-model-for-video-prediction","title":"Eidetic 3D LSTM: A Model for Video Prediction and Beyond","date":"2019-05-01","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/memory-in-memory-a-predictive-neural-network","title":"Memory In Memory: A Predictive Neural Network for Learning Higher-Order Non-Stationarity from Spatiotemporal Dynamics","date":"2018-11-19","rows_on_this_dataset":2,"code_links":4,"syntology":null},{"paper":"/paper/predrnn-towards-a-resolution-of-the-deep-in","title":"PredRNN++: Towards A Resolution of the Deep-in-Time Dilemma in Spatiotemporal Predictive Learning","date":"2018-04-17","rows_on_this_dataset":1,"code_links":11,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":0,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/predrnn-recurrent-neural-networks-for-1","title":"PredRNN: Recurrent Neural Networks for Predictive Learning using Spatiotemporal LSTMs","date":"2017-12-01","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/convolutional-lstm-network-a-machine-learning","title":"Convolutional LSTM Network: A Machine Learning Approach for Precipitation Nowcasting","date":"2015-06-13","rows_on_this_dataset":1,"code_links":23,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":1,"samples_unverified":2,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":6,"samples_harvested":36,"samples_ran":19,"samples_unverified":17,"pointer_only_for_licence":12,"papers_with_no_sample_that_ran":3,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}