{"url":"/dataset/d4rl","name":"D4RL","full_name":"D4RL","description_markdown":"**D4RL** is a collection of environments for offline reinforcement learning. These environments include Maze2D, AntMaze, Adroit, Gym, Flow, FrankKitchen and CARLA.\r\n\r\nSource: [https://sites.google.com/view/d4rl/home](https://sites.google.com/view/d4rl/home)\r\nImage Source: [https://sites.google.com/view/d4rl/home](https://sites.google.com/view/d4rl/home)","description_withheld":null,"homepage":"https://sites.google.com/view/d4rl/home","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/datasets-for-data-driven-reinforcement","title":"D4RL: Datasets for Deep Data-Driven Reinforcement Learning","first_author":"Justin Fu","url":null},"license":{"name":"Apache-2 / CC-BY","url":"https://github.com/rail-berkeley/d4rl"},"modalities":[{"name":"Environment","url":"/datasets/modality/environment"}],"tasks":[{"name":"Continuous Control","url":"/task/continuous-control","datasets_with_task":"/datasets/task/continuous-control"},{"name":"Decision Making","url":"/task/decision-making","datasets_with_task":"/datasets/task/decision-making"},{"name":"Offline RL","url":"/task/offline-rl","datasets_with_task":"/datasets/task/offline-rl"},{"name":"D4RL","url":"/task/d4rl","datasets_with_task":"/datasets/task/d4rl"},{"name":"Gym halfcheetah-random","url":"/task/gym-halfcheetah-random","datasets_with_task":"/datasets/task/gym-halfcheetah-random"},{"name":"Gym halfcheetah-medium","url":"/task/gym-halfcheetah-medium","datasets_with_task":"/datasets/task/gym-halfcheetah-medium"},{"name":"Gym halfcheetah-expert","url":"/task/gym-halfcheetah-expert","datasets_with_task":"/datasets/task/gym-halfcheetah-expert"},{"name":"Gym halfcheetah-medium-expert","url":"/task/gym-halfcheetah-medium-expert","datasets_with_task":"/datasets/task/gym-halfcheetah-medium-expert"},{"name":"Gym halfcheetah-medium-replay","url":"/task/gym-halfcheetah-medium-replay","datasets_with_task":"/datasets/task/gym-halfcheetah-medium-replay"},{"name":"Gym halfcheetah-full-replay","url":"/task/gym-halfcheetah-full-replay","datasets_with_task":"/datasets/task/gym-halfcheetah-full-replay"},{"name":"Adroid pen-human","url":"/task/adroid-pen-human","datasets_with_task":"/datasets/task/adroid-pen-human"},{"name":"Adroid hammer-human","url":"/task/adroid-hammer-human","datasets_with_task":"/datasets/task/adroid-hammer-human"},{"name":"Adroid door-human","url":"/task/adroid-door-human","datasets_with_task":"/datasets/task/adroid-door-human"},{"name":"Adroid relocate-human","url":"/task/adroid-relocate-human","datasets_with_task":"/datasets/task/adroid-relocate-human"},{"name":"Adroid pen-cloned","url":"/task/adroid-pen-cloned","datasets_with_task":"/datasets/task/adroid-pen-cloned"},{"name":"Adroid hammer-cloned","url":"/task/adroid-hammer-cloned","datasets_with_task":"/datasets/task/adroid-hammer-cloned"},{"name":"Adroid door-cloned","url":"/task/adroid-door-cloned","datasets_with_task":"/datasets/task/adroid-door-cloned"},{"name":"Adroid relocate-cloned","url":"/task/adroid-relocate-cloned","datasets_with_task":"/datasets/task/adroid-relocate-cloned"}],"languages":[],"variants":["D4RL"],"data_loaders":[{"repo":"https://github.com/tensorflow/datasets","url":"https://www.tensorflow.org/datasets/catalog/d4rl_mujoco_walker2d","frameworks":["tf","jax"]},{"repo":"https://github.com/tensorflow/datasets","url":"https://www.tensorflow.org/datasets/catalog/d4rl_mujoco_hopper","frameworks":["tf","jax"]},{"repo":"https://github.com/tensorflow/datasets","url":"https://www.tensorflow.org/datasets/catalog/d4rl_mujoco_halfcheetah","frameworks":["tf","jax"]},{"repo":"https://github.com/tensorflow/datasets","url":"https://www.tensorflow.org/datasets/catalog/d4rl_mujoco_ant","frameworks":["tf","jax"]}],"num_papers_in_archive":538,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/d4rl-on-d4rl","task":"D4RL","dataset_variant":"D4RL","rows":9,"metrics":["Average Reward"],"first_row_in_archive_order":{"model":"PMDB","paper":"/paper/model-based-offline-reinforcement-learning","metrics":{"Average Reward":"88.2"},"code_links":[{"title":"huawei-noah/HEBO","url":"https://github.com/huawei-noah/HEBO/tree/master/PMDB"},{"title":"huawei-noah/hebo","url":"https://github.com/huawei-noah/hebo"},{"title":"2023-MindSpore-1/ms-code-220","url":"https://github.com/2023-MindSpore-1/ms-code-220/tree/main/pmdb"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/offline-rl-on-d4rl","task":"Offline RL","dataset_variant":"D4RL","rows":3,"metrics":["Average Reward"],"first_row_in_archive_order":{"model":"KFC","paper":"/paper/koopman-q-learning-offline-reinforcement-1","metrics":{"Average Reward":"81.8"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/any-step-dynamics-model-improves-future","title":"Any-step Dynamics Model Improves Future Predictions for Online and Offline Reinforcement Learning","date":"2024-05-27","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/primal-attention-self-attention-through","title":"Primal-Attention: Self-attention through Asymmetric Kernel SVD in Primal Representation","date":"2023-05-31","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/model-based-offline-reinforcement-learning","title":"Model-Based Offline Reinforcement Learning with Pessimism-Modulated Dynamics Belief","date":"2022-10-13","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/cosformer-rethinking-softmax-in-attention-1","title":"cosFormer: Rethinking Softmax in Attention","date":"2022-02-17","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/flowformer-linearizing-transformers-with","title":"Flowformer: Linearizing Transformers with Conservation Flows","date":"2022-02-13","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":3,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/koopman-q-learning-offline-reinforcement-1","title":"Koopman Q-learning: Offline Reinforcement Learning via Symmetries of Dynamics","date":"2021-11-02","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/decision-transformer-reinforcement-learning","title":"Decision Transformer: Reinforcement Learning via Sequence Modeling","date":"2021-06-02","rows_on_this_dataset":2,"code_links":20,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":26,"samples_ran":17,"samples_unverified":9,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/rethinking-attention-with-performers","title":"Rethinking Attention with Performers","date":"2020-09-30","rows_on_this_dataset":1,"code_links":7,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":16,"samples_ran":9,"samples_unverified":7,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/transformers-are-rnns-fast-autoregressive","title":"Transformers are RNNs: Fast Autoregressive Transformers with Linear Attention","date":"2020-06-29","rows_on_this_dataset":1,"code_links":8,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":3,"samples_unverified":5,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/reformer-the-efficient-transformer-1","title":"Reformer: The Efficient Transformer","date":"2020-01-13","rows_on_this_dataset":1,"code_links":10,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":6,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":7,"samples_harvested":66,"samples_ran":40,"samples_unverified":26,"pointer_only_for_licence":16,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}