{"url":"/dataset/bair-robot-pushing","name":"BAIR Robot Pushing","full_name":null,"description_markdown":"Dataset of 64x64 images of a robot pushing objects on a table top. From Berkeley AI Research (BAIR).\r\n\r\nSource: Self-Supervised Visual Planning with Temporal Skip Connections (https://arxiv.org/abs/1710.05268)\r\n\r\nVideo prediction : Conditioned on 2 frames, predict 14 frames.","description_withheld":null,"homepage":"https://sites.google.com/berkeley.edu/robotic-interaction-datasets","introduced_date":null,"introduced_date_note":null,"introduced_by":null,"license":null,"modalities":[],"tasks":[{"name":"Video Prediction","url":"/task/video-prediction","datasets_with_task":"/datasets/task/video-prediction"},{"name":"Video Generation","url":"/task/video-generation","datasets_with_task":"/datasets/task/video-generation"}],"languages":[],"variants":["BAIR Robot Pushing"],"data_loaders":[],"num_papers_in_archive":27,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/video-generation-on-bair-robot-pushing","task":"Video Generation","dataset_variant":"BAIR Robot Pushing","rows":31,"metrics":["FVD score","SSIM","PSNR","LPIPS","Cond","Train","Pred","Notes"],"first_row_in_archive_order":{"model":"MAGVIT","paper":"/paper/magvit-masked-generative-video-transformer","metrics":{"Cond":"1","FVD score":"62","Pred":"15","Train":"15"},"code_links":[{"title":"google-research/magvit","url":"https://github.com/google-research/magvit"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/video-prediction-on-bair-robot-pushing-1","task":"Video Prediction","dataset_variant":"BAIR Robot Pushing","rows":6,"metrics":["FVD"],"first_row_in_archive_order":{"model":"MAGVIT (-L-FP)","paper":"/paper/magvit-masked-generative-video-transformer","metrics":{"FVD":"62±0.1"},"code_links":[{"title":"google-research/magvit","url":"https://github.com/google-research/magvit"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/magvit-masked-generative-video-transformer","title":"MAGVIT: Masked Generative Video Transformer","date":"2022-12-10","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":1,"samples_unverified":9,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/tell-me-what-happened-unifying-text-guided","title":"Tell Me What Happened: Unifying Text-guided Video Completion via Multimodal Masked Video Generation","date":"2022-11-23","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/phenaki-variable-length-video-generation-from","title":"Phenaki: Variable Length Video Generation From Open Domain Textual Description","date":"2022-10-05","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/diffusion-models-for-video-prediction-and","title":"Diffusion Models for Video Prediction and Infilling","date":"2022-06-15","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":19,"samples_ran":9,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/masked-conditional-video-diffusion-for","title":"MCVD: Masked Conditional Video Diffusion for Prediction, Generation, and Interpolation","date":"2022-05-19","rows_on_this_dataset":3,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":16,"samples_ran":7,"samples_unverified":9,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/nuwa-visual-synthesis-pre-training-for-neural","title":"NÜWA: Visual Synthesis Pre-training for Neural visUal World creAtion","date":"2021-11-24","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/slamp-stochastic-latent-appearance-and-motion","title":"SLAMP: Stochastic Latent Appearance and Motion Prediction","date":"2021-08-05","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/ccvs-context-aware-controllable-video","title":"CCVS: Context-aware Controllable Video Synthesis","date":"2021-07-16","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":3,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/diverse-video-generation-using-a-gaussian-1","title":"Diverse Video Generation using a Gaussian Process Trigger","date":"2021-07-09","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/fitvid-overfitting-in-pixel-level-video","title":"FitVid: Overfitting in Pixel-Level Video Prediction","date":"2021-06-24","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":0,"samples_unverified":11,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/videogpt-video-generation-using-vq-vae-and","title":"VideoGPT: Video Generation using VQ-VAE and Transformers","date":"2021-04-20","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/latent-video-transformer","title":"Latent Video Transformer","date":"2020-06-18","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":1,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/transformation-based-adversarial-video","title":"Transformation-based Adversarial Video Prediction on Large-Scale Data","date":"2020-03-09","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/exploring-spatial-temporal-multi-frequency","title":"Exploring Spatial-Temporal Multi-Frequency Analysis for High-Fidelity and Temporal-Consistency Video Prediction","date":"2020-02-23","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/stochastic-latent-residual-video-prediction-1","title":"Stochastic Latent Residual Video Prediction","date":"2020-02-21","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":2,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/efficient-video-generation-on-complex","title":"Adversarial Video Generation on Complex Datasets","date":"2019-07-15","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/scaling-autoregressive-video-models","title":"Scaling Autoregressive Video Models","date":"2019-06-06","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/improved-conditional-vrnns-for-video","title":"Improved Conditional VRNNs for Video Prediction","date":"2019-04-27","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/videoflow-a-flow-based-generative-model-for","title":"VideoFlow: A Conditional Flow-Based Model for Stochastic Video Generation","date":"2019-03-04","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/stochastic-adversarial-video-prediction","title":"Stochastic Adversarial Video Prediction","date":"2018-04-04","rows_on_this_dataset":4,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":16,"samples_ran":4,"samples_unverified":12,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/stochastic-video-generation-with-a-learned","title":"Stochastic Video Generation with a Learned Prior","date":"2018-02-21","rows_on_this_dataset":3,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/stochastic-variational-video-prediction","title":"Stochastic Variational Video Prediction","date":"2017-10-30","rows_on_this_dataset":2,"code_links":3,"syntology":null},{"paper":"/paper/mocogan-decomposing-motion-and-content-for","title":"MoCoGAN: Decomposing Motion and Content for Video Generation","date":"2017-07-17","rows_on_this_dataset":1,"code_links":5,"syntology":null},{"paper":"/paper/unsupervised-learning-for-physical","title":"Unsupervised Learning for Physical Interaction through Video Prediction","date":"2016-05-23","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":13,"samples_harvested":109,"samples_ran":38,"samples_unverified":71,"pointer_only_for_licence":11,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}