{"url":"/task/d4rl","name":"D4RL","slug":"d4rl","description_markdown":null,"categories":[{"name":"Robots","url":"/area/robots"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":226,"papers_with_code":108,"benchmarks":1,"benchmark_tables_in_archive":1,"benchmark_tables_shown":1,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":1,"subtasks":0,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/d4rl-on-d4rl","slug":"d4rl-on-d4rl","dataset":"D4RL","dataset_url":"/dataset/d4rl","rows_in_archive":9,"metrics":["Average Reward"],"first_row_in_archive_order":{"model":"PMDB","paper_title":"Model-Based Offline Reinforcement Learning with Pessimism-Modulated Dynamics Belief","paper_url":"/paper/model-based-offline-reinforcement-learning","paper_date":"2022-10-13","arxiv_id":"2210.06692","code_links":[{"title":"huawei-noah/HEBO","url":"https://github.com/huawei-noah/HEBO/tree/master/PMDB"},{"title":"huawei-noah/hebo","url":"https://github.com/huawei-noah/hebo"},{"title":"2023-MindSpore-1/ms-code-220","url":"https://github.com/2023-MindSpore-1/ms-code-220/tree/main/pmdb"}],"syntology":null}}],"datasets":[{"url":"/dataset/d4rl","name":"D4RL","full_name":"D4RL","num_papers_in_archive":538}],"subtasks":[],"parent_tasks":[{"url":"/task/mujoco-games","name":"MuJoCo Games"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":108,"tagged_in_all":226,"items":[{"url":"/paper/decision-transformer-reinforcement-learning","title":"Decision Transformer: Reinforcement Learning via Sequence Modeling","date":"2021-06-02","arxiv_id":"2106.01345","repositories_listed":20,"syntology":{"n":26,"n_ran":17,"n_unverified":9,"n_pointer_only":6}},{"url":"/paper/offline-reinforcement-learning-with-implicit","title":"Offline Reinforcement Learning with Implicit Q-Learning","date":"2021-10-12","arxiv_id":"2110.06169","repositories_listed":17,"syntology":{"n":58,"n_ran":32,"n_unverified":26,"n_pointer_only":22}},{"url":"/paper/reformer-the-efficient-transformer-1","title":"Reformer: The Efficient Transformer","date":"2020-01-13","arxiv_id":"2001.04451","repositories_listed":10,"syntology":{"n":8,"n_ran":6,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/transformers-are-rnns-fast-autoregressive","title":"Transformers are RNNs: Fast Autoregressive Transformers with Linear Attention","date":"2020-06-29","arxiv_id":"2006.16236","repositories_listed":8,"syntology":{"n":8,"n_ran":3,"n_unverified":5,"n_pointer_only":3}},{"url":"/paper/rethinking-attention-with-performers","title":"Rethinking Attention with Performers","date":"2020-09-30","arxiv_id":"2009.14794","repositories_listed":7,"syntology":{"n":16,"n_ran":9,"n_unverified":7,"n_pointer_only":6}},{"url":"/paper/datasets-for-data-driven-reinforcement","title":"D4RL: Datasets for Deep Data-Driven Reinforcement Learning","date":"2020-04-15","arxiv_id":"2004.07219","repositories_listed":7,"syntology":{"n":12,"n_ran":4,"n_unverified":8,"n_pointer_only":0}},{"url":"/paper/corl-research-oriented-deep-offline-1","title":"CORL: Research-oriented Deep Offline Reinforcement Learning Library","date":"2022-10-13","arxiv_id":"2210.07105","repositories_listed":5,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/uncertainty-based-offline-reinforcement","title":"Uncertainty-Based Offline Reinforcement Learning with Diversified Q-Ensemble","date":"2021-10-04","arxiv_id":"2110.01548","repositories_listed":5,"syntology":{"n":21,"n_ran":13,"n_unverified":8,"n_pointer_only":6}},{"url":"/paper/implicit-behavioral-cloning","title":"Implicit Behavioral Cloning","date":"2021-09-01","arxiv_id":"2109.00137","repositories_listed":5,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/offline-rl-with-no-ood-actions-in-sample","title":"Offline RL with No OOD Actions: In-Sample Learning via Implicit Value Regularization","date":"2023-03-28","arxiv_id":"2303.15810","repositories_listed":4,"syntology":{"n":3,"n_ran":1,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/extreme-q-learning-maxent-rl-without-entropy","title":"Extreme Q-Learning: MaxEnt RL without Entropy","date":"2023-01-05","arxiv_id":"2301.02328","repositories_listed":4,"syntology":{"n":13,"n_ran":8,"n_unverified":5,"n_pointer_only":7}},{"url":"/paper/datasets-and-benchmarks-for-offline-safe","title":"Datasets and Benchmarks for Offline Safe Reinforcement Learning","date":"2023-06-15","arxiv_id":"2306.09303","repositories_listed":3,"syntology":{"n":3,"n_ran":0,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/revisiting-the-minimalist-approach-to-offline","title":"Revisiting the Minimalist Approach to Offline Reinforcement Learning","date":"2023-05-16","arxiv_id":"2305.09836","repositories_listed":3,"syntology":{"n":16,"n_ran":7,"n_unverified":9,"n_pointer_only":0}},{"url":"/paper/contrastive-energy-prediction-for-exact","title":"Contrastive Energy Prediction for Exact Energy-Guided Diffusion Sampling in Offline Reinforcement Learning","date":"2023-04-25","arxiv_id":"2304.12824","repositories_listed":3,"syntology":{"n":27,"n_ran":17,"n_unverified":10,"n_pointer_only":5}},{"url":"/paper/anti-exploration-by-random-network","title":"Anti-Exploration by Random Network Distillation","date":"2023-01-31","arxiv_id":"2301.13616","repositories_listed":3,"syntology":{"n":26,"n_ran":20,"n_unverified":6,"n_pointer_only":7}},{"url":"/paper/model-based-offline-reinforcement-learning","title":"Model-Based Offline Reinforcement Learning with Pessimism-Modulated Dynamics Belief","date":"2022-10-13","arxiv_id":"2210.06692","repositories_listed":3,"syntology":null},{"url":"/paper/diffusion-policies-as-an-expressive-policy","title":"Diffusion Policies as an Expressive Policy Class for Offline Reinforcement Learning","date":"2022-08-12","arxiv_id":"2208.06193","repositories_listed":3,"syntology":{"n":18,"n_ran":11,"n_unverified":7,"n_pointer_only":10}},{"url":"/paper/mildly-conservative-q-learning-for-offline","title":"Mildly Conservative Q-Learning for Offline Reinforcement Learning","date":"2022-06-09","arxiv_id":"2206.04745","repositories_listed":3,"syntology":{"n":16,"n_ran":12,"n_unverified":4,"n_pointer_only":6}},{"url":"/paper/cosformer-rethinking-softmax-in-attention-1","title":"cosFormer: Rethinking Softmax in Attention","date":"2022-02-17","arxiv_id":"2202.08791","repositories_listed":3,"syntology":null},{"url":"/paper/adversarially-trained-actor-critic-for","title":"Adversarially Trained Actor Critic for Offline Reinforcement Learning","date":"2022-02-05","arxiv_id":"2202.02446","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/flow-q-learning","title":"Flow Q-Learning","date":"2025-02-04","arxiv_id":"2502.02538","repositories_listed":2,"syntology":{"n":7,"n_ran":5,"n_unverified":2,"n_pointer_only":5}},{"url":"/paper/q-value-regularized-transformer-for-offline","title":"Q-value Regularized Transformer for Offline Reinforcement Learning","date":"2024-05-27","arxiv_id":"2405.17098","repositories_listed":2,"syntology":{"n":8,"n_ran":2,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/hierarchical-decision-mamba","title":"Decision Mamba Architectures","date":"2024-05-13","arxiv_id":"2405.07943","repositories_listed":2,"syntology":null},{"url":"/paper/exploration-and-anti-exploration-with","title":"Exploration and Anti-Exploration with Distributional Random Network Distillation","date":"2024-01-18","arxiv_id":"2401.09750","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/model-bellman-inconsistency-for-model-based","title":"Model-Bellman Inconsistency for Model-based Offline Reinforcement Learning","date":"2023-07-01","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/behavior-proximal-policy-optimization","title":"Behavior Proximal Policy Optimization","date":"2023-02-22","arxiv_id":"2302.11312","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/understanding-expertise-through","title":"When Demonstrations Meet Generative World Models: A Maximum Likelihood Framework for Offline Inverse Reinforcement Learning","date":"2023-02-15","arxiv_id":"2302.07457","repositories_listed":2,"syntology":null},{"url":"/paper/adaptive-behavior-cloning-regularization-for-1","title":"Adaptive Behavior Cloning Regularization for Stable Offline-to-Online Reinforcement Learning","date":"2022-10-25","arxiv_id":"2210.13846","repositories_listed":2,"syntology":{"n":3,"n_ran":0,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/why-so-pessimistic-estimating-uncertainties-1","title":"Why So Pessimistic? Estimating Uncertainties for Offline RL through Ensembles, and Why Their Independence Matters","date":"2022-05-27","arxiv_id":"2205.13703","repositories_listed":2,"syntology":null},{"url":"/paper/distance-sensitive-offline-reinforcement","title":"When Data Geometry Meets Deep Function: Generalizing Offline Reinforcement Learning","date":"2022-05-23","arxiv_id":"2205.11027","repositories_listed":2,"syntology":null}],"syntology_records":23,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}