{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/atari-games/papers/4","list_of":"/task/atari-games","task":"Atari Games","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":4,"pages_in_order":7,"rows_per_page":100,"rows":[301,400],"of":625,"counts":{"archive_papers_tagged":625,"with_a_code_link":313,"where_syntology_ran_a_sample":118,"not_listed_spam_title":0,"listed":625,"listed_where_code_ran":118,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":97,"every_run_a_failure_of_syntologys_instrument":21,"listed_with_a_run_with_no_instrument_failure":97,"listed_every_run_a_failure_of_syntologys_instrument":21,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/atari-games","prev":"/task/atari-games/papers/3","next":"/task/atari-games/papers/5","papers":[{"url":"/paper/deep-reinforcement-learning-framework-for","slug":"deep-reinforcement-learning-framework-for","title":"Deep Reinforcement Learning framework for Autonomous Driving","date":"2017-04-08","arxiv_id":"1704.02532","repositories_listed":1,"syntology":null},{"url":"/paper/recurrent-environment-simulators","slug":"recurrent-environment-simulators","title":"Recurrent Environment Simulators","date":"2017-04-07","arxiv_id":"1704.02254","repositories_listed":1,"syntology":null},{"url":"/paper/count-based-exploration-with-neural-density","slug":"count-based-exploration-with-neural-density","title":"Count-Based Exploration with Neural Density Models","date":"2017-03-03","arxiv_id":"1703.01310","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/count-based-exploration-with-neural-density#ran","syntology_url":"https://syntology.ai/paper/1703.01310","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1703.01310"}},"official":null}},{"url":"/paper/a-laplacian-framework-for-option-discovery-in","slug":"a-laplacian-framework-for-option-discovery-in","title":"A Laplacian Framework for Option Discovery in Reinforcement Learning","date":"2017-03-02","arxiv_id":"1703.00956","repositories_listed":1,"syntology":null},{"url":"/paper/beating-the-worlds-best-at-super-smash-bros","slug":"beating-the-worlds-best-at-super-smash-bros","title":"Beating the World's Best at Super Smash Bros. with Deep Reinforcement Learning","date":"2017-02-21","arxiv_id":"1702.06230","repositories_listed":1,"syntology":null},{"url":"/paper/playing-snes-in-the-retro-learning","slug":"playing-snes-in-the-retro-learning","title":"Playing SNES in the Retro Learning Environment","date":"2016-11-07","arxiv_id":"1611.02205","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-play-in-a-day-faster-deep","slug":"learning-to-play-in-a-day-faster-deep","title":"Learning to Play in a Day: Faster Deep Reinforcement Learning by Optimality Tightening","date":"2016-11-05","arxiv_id":"1611.01606","repositories_listed":1,"syntology":null},{"url":"/paper/playing-atari-games-with-deep-reinforcement","slug":"playing-atari-games-with-deep-reinforcement","title":"Playing Atari Games with Deep Reinforcement Learning and Human Checkpoint Replay","date":"2016-07-18","arxiv_id":"1607.05077","repositories_listed":1,"syntology":null},{"url":"/paper/unifying-count-based-exploration-and","slug":"unifying-count-based-exploration-and","title":"Unifying Count-Based Exploration and Intrinsic Motivation","date":"2016-06-06","arxiv_id":"1606.01868","repositories_listed":1,"syntology":null},{"url":"/paper/true-online-temporal-difference-learning","slug":"true-online-temporal-difference-learning","title":"True Online Temporal-Difference Learning","date":"2015-12-13","arxiv_id":"1512.04087","repositories_listed":1,"syntology":null},{"url":"/paper/state-of-the-art-control-of-atari-games-using","slug":"state-of-the-art-control-of-atari-games-using","title":"State of the Art Control of Atari Games Using Shallow Reinforcement Learning","date":"2015-12-04","arxiv_id":"1512.01563","repositories_listed":1,"syntology":null},{"url":"/paper/action-conditional-video-prediction-using","slug":"action-conditional-video-prediction-using","title":"Action-Conditional Video Prediction using Deep Networks in Atari Games","date":"2015-07-31","arxiv_id":"1507.08750","repositories_listed":1,"syntology":null},{"url":"/paper/incentivizing-exploration-in-reinforcement","slug":"incentivizing-exploration-in-reinforcement","title":"Incentivizing Exploration In Reinforcement Learning With Deep Predictive Models","date":"2015-07-03","arxiv_id":"1507.00814","repositories_listed":1,"syntology":null},{"url":null,"slug":"a-principled-path-to-fitted-distributional","title":"A Principled Path to Fitted Distributional Evaluation","date":"2025-06-24","arxiv_id":"2506.20048","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-performance-of-spike-based-deep-q","title":"Improving Performance of Spike-based Deep Q-Learning using Ternary Neurons","date":"2025-06-03","arxiv_id":"2506.03392","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-11478","title":"Automatic Reward Shaping from Confounded Offline Data","date":"2025-05-16","arxiv_id":"2505.11478","repositories_listed":0,"syntology":null},{"url":null,"slug":"switchmt-an-adaptive-context-switching","title":"SwitchMT: An Adaptive Context Switching Methodology for Scalable Multi-Task Learning in Intelligent Autonomous Agents","date":"2025-04-18","arxiv_id":"2504.13541","repositories_listed":0,"syntology":null},{"url":null,"slug":"look-before-leap-look-ahead-planning-with","title":"Look Before Leap: Look-Ahead Planning with Uncertainty in Reinforcement Learning","date":"2025-03-26","arxiv_id":"2503.20139","repositories_listed":0,"syntology":null},{"url":null,"slug":"adventurer-exploration-with-bigan-for-deep","title":"Adventurer: Exploration with BiGAN for Deep Reinforcement Learning","date":"2025-03-24","arxiv_id":"2503.18612","repositories_listed":0,"syntology":null},{"url":null,"slug":"apf-boosting-adaptive-potential-function","title":"APF+: Boosting adaptive-potential function reinforcement learning methods with a W-shaped network for high-dimensional games","date":"2025-03-17","arxiv_id":"2503.13557","repositories_listed":0,"syntology":null},{"url":null,"slug":"target-return-optimizer-for-multi-game","title":"Target Return Optimizer for Multi-Game Decision Transformer","date":"2025-03-04","arxiv_id":"2503.02311","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-explore-with-predictive-world","title":"Learning To Explore With Predictive World Model Via Self-Supervised Learning","date":"2025-02-18","arxiv_id":"2502.13200","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-strategy-based-and","title":"Reinforcement Learning in Strategy-Based and Atari Games: A Review of Google DeepMinds Innovations","date":"2025-02-14","arxiv_id":"2502.10303","repositories_listed":0,"syntology":null},{"url":null,"slug":"activation-by-interval-wise-dropout-a-simple","title":"Activation by Interval-wise Dropout: A Simple Way to Prevent Neural Networks from Plasticity Loss","date":"2025-02-03","arxiv_id":"2502.01342","repositories_listed":0,"syntology":null},{"url":null,"slug":"objects-matter-object-centric-world-models","title":"Objects matter: object-centric world models improve reinforcement learning in visually complex environments","date":"2025-01-27","arxiv_id":"2501.16443","repositories_listed":0,"syntology":null},{"url":null,"slug":"group-agent-reinforcement-learning-with","title":"Group-Agent Reinforcement Learning with Heterogeneous Agents","date":"2025-01-21","arxiv_id":"2501.11818","repositories_listed":0,"syntology":null},{"url":null,"slug":"mtspark-enabling-multi-task-learning-with","title":"MTSpark: Enabling Multi-Task Learning with Spiking Neural Networks for Generalist Agents","date":"2024-12-06","arxiv_id":"2412.04847","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-code-to-play-benchmarking-program-search","title":"From Code to Play: Benchmarking Program Search for Games Using Large Language Models","date":"2024-12-05","arxiv_id":"2412.04057","repositories_listed":0,"syntology":null},{"url":null,"slug":"why-the-agent-made-that-decision-explaining","title":"Why the Agent Made that Decision: Explaining Deep Reinforcement Learning with Vision Masks","date":"2024-11-25","arxiv_id":"2411.16120","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpreting-the-learned-model-in-muzero","title":"Interpreting the Learned Model in MuZero Planning","date":"2024-11-07","arxiv_id":"2411.04580","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-diversity-based-experience-replay","title":"Efficient Diversity-based Experience Replay for Deep Reinforcement Learning","date":"2024-10-27","arxiv_id":"2410.20487","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-end-to-end-neurosymbolic","title":"Interpretable end-to-end Neurosymbolic Reinforcement Learning agents","date":"2024-10-18","arxiv_id":"2410.14371","repositories_listed":0,"syntology":null},{"url":null,"slug":"transforming-game-play-a-comparative-study-of","title":"Transforming Game Play: A Comparative Study of DCQN and DTQN Architectures in Reinforcement Learning","date":"2024-10-14","arxiv_id":"2410.10660","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-from-imperfect-1","title":"Reinforcement Learning From Imperfect Corrective Actions And Proxy Rewards","date":"2024-10-08","arxiv_id":"2410.05782","repositories_listed":0,"syntology":null},{"url":null,"slug":"atari-gpt-investigating-the-capabilities-of","title":"Atari-GPT: Benchmarking Multimodal Large Language Models as Low-Level Policies in Atari Games","date":"2024-08-28","arxiv_id":"2408.15950","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-generalizable-reinforcement-learning-1","title":"Towards Generalizable Reinforcement Learning via Causality-Guided Self-Adaptive Representations","date":"2024-07-30","arxiv_id":"2407.20651","repositories_listed":0,"syntology":null},{"url":null,"slug":"pg-rainbow-using-distributional-reinforcement","title":"PG-Rainbow: Using Distributional Reinforcement Learning in Policy Gradient Methods","date":"2024-07-18","arxiv_id":"2407.13146","repositories_listed":0,"syntology":null},{"url":null,"slug":"normalization-and-effective-learning-rates-in","title":"Normalization and effective learning rates in reinforcement learning","date":"2024-07-01","arxiv_id":"2407.01800","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-and-diagnosing-deep","title":"Understanding and Diagnosing Deep Reinforcement Learning","date":"2024-06-23","arxiv_id":"2406.16979","repositories_listed":0,"syntology":null},{"url":"/paper/adaptive-q-network-on-the-fly-target","slug":"adaptive-q-network-on-the-fly-target","title":"Adaptive $Q$-Network: On-the-fly Target Selection for Deep Reinforcement Learning","date":"2024-05-25","arxiv_id":"2405.16195","repositories_listed":0,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/adaptive-q-network-on-the-fly-target#ran","syntology_url":"https://syntology.ai/paper/2405.16195","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.16195"}},"official":null}},{"url":null,"slug":"utilizing-maximum-mean-discrepancy-barycenter","title":"Utilizing Maximum Mean Discrepancy Barycenter for Propagating the Uncertainty of Value Functions in Reinforcement Learning","date":"2024-03-31","arxiv_id":"2404.00686","repositories_listed":0,"syntology":null},{"url":null,"slug":"stop-regressing-training-value-functions-via","title":"Stop Regressing: Training Value Functions via Classification for Scalable Deep RL","date":"2024-03-06","arxiv_id":"2403.03950","repositories_listed":0,"syntology":null},{"url":null,"slug":"iterated-q-network-beyond-the-one-step","title":"Iterated $Q$-Network: Beyond One-Step Bellman Updates in Deep Reinforcement Learning","date":"2024-03-04","arxiv_id":"2403.02107","repositories_listed":0,"syntology":null},{"url":null,"slug":"disentangling-the-causes-of-plasticity-loss","title":"Disentangling the Causes of Plasticity Loss in Neural Networks","date":"2024-02-29","arxiv_id":"2402.18762","repositories_listed":0,"syntology":null},{"url":null,"slug":"control-in-stochastic-environment-with-delays","title":"Control in Stochastic Environment with Delays: A Model-based Reinforcement Learning Approach","date":"2024-02-01","arxiv_id":"2402.00313","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-policy-style-transfer","title":"Neural Policy Style Transfer","date":"2024-02-01","arxiv_id":"2402.00677","repositories_listed":0,"syntology":null},{"url":null,"slug":"look-around-unexpected-gains-from-training-on","title":"The Indoor-Training Effect: unexpected gains from distribution shifts in the transition function","date":"2024-01-29","arxiv_id":"2401.15856","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-encoders-for-data-efficient-imitation","title":"Visual Encoders for Data-Efficient Imitation Learning in Modern Video Games","date":"2023-12-04","arxiv_id":"2312.02312","repositories_listed":0,"syntology":null},{"url":null,"slug":"agent-aware-training-for-agent-agnostic","title":"Agent-Aware Training for Agent-Agnostic Action Advising in Deep Reinforcement Learning","date":"2023-11-28","arxiv_id":"2311.16807","repositories_listed":0,"syntology":null},{"url":null,"slug":"minimax-exploiter-a-data-efficient-approach","title":"Minimax Exploiter: A Data Efficient Approach for Competitive Self-Play","date":"2023-11-28","arxiv_id":"2311.17190","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-pretrained-models-for-deployable","title":"Evaluating Pretrained models for Deployable Lifelong Learning","date":"2023-11-22","arxiv_id":"2311.13648","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-what-to-when-a-spiking-neural-network","title":"From \"What\" to \"When\" -- a Spiking Neural Network Predicting Rare Events and Time to their Occurrence","date":"2023-11-09","arxiv_id":"2311.05210","repositories_listed":0,"syntology":null},{"url":null,"slug":"dsac-c-constrained-maximum-entropy-for-robust","title":"DSAC-C: Constrained Maximum Entropy for Robust Discrete Soft-Actor Critic","date":"2023-10-26","arxiv_id":"2310.17173","repositories_listed":0,"syntology":null},{"url":"/paper/reward-scale-robustness-for-proximal-policy","slug":"reward-scale-robustness-for-proximal-policy","title":"Reward Scale Robustness for Proximal Policy Optimization via DreamerV3 Tricks","date":"2023-10-26","arxiv_id":"2310.17805","repositories_listed":0,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reward-scale-robustness-for-proximal-policy#ran","syntology_url":"https://syntology.ai/paper/2310.17805","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.17805"}},"official":null}},{"url":null,"slug":"towards-control-centric-representations-in","title":"Towards Control-Centric Representations in Reinforcement Learning from Images","date":"2023-10-25","arxiv_id":"2310.16655","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-actions-and-control-of-focus-of","title":"Learning Actions and Control of Focus of Attention with a Log-Polar-like Sensor","date":"2023-09-22","arxiv_id":"2309.12634","repositories_listed":0,"syntology":null},{"url":null,"slug":"soft-decomposed-policy-critic-bridging-the","title":"Soft Decomposed Policy-Critic: Bridging the Gap for Effective Continuous Control with Discrete RL","date":"2023-08-20","arxiv_id":"2308.10203","repositories_listed":0,"syntology":null},{"url":null,"slug":"bag-of-policies-for-distributional-deep","title":"Bag of Policies for Distributional Deep Exploration","date":"2023-08-03","arxiv_id":"2308.01759","repositories_listed":0,"syntology":null},{"url":null,"slug":"elastic-decision-transformer-1","title":"Elastic Decision Transformer","date":"2023-07-05","arxiv_id":"2307.02484","repositories_listed":0,"syntology":null},{"url":null,"slug":"action-q-transformer-visual-explanation-in","title":"Action Q-Transformer: Visual Explanation in Deep Reinforcement Learning with Encoder-Decoder Model using Action Query","date":"2023-06-24","arxiv_id":"2306.13879","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-rl-perceptron-generalisation-dynamics-of","title":"The RL Perceptron: Generalisation Dynamics of Policy Learning in High Dimensions","date":"2023-06-17","arxiv_id":"2306.10404","repositories_listed":0,"syntology":null},{"url":null,"slug":"vanishing-bias-heuristic-guided-reinforcement","title":"Vanishing Bias Heuristic-guided Reinforcement Learning Algorithm","date":"2023-06-17","arxiv_id":"2306.10216","repositories_listed":0,"syntology":null},{"url":null,"slug":"detecting-adversarial-directions-in-deep","title":"Detecting Adversarial Directions in Deep Reinforcement Learning to Make Robust Decisions","date":"2023-06-09","arxiv_id":"2306.05873","repositories_listed":0,"syntology":null},{"url":null,"slug":"successor-predecessor-intrinsic-exploration","title":"Successor-Predecessor Intrinsic Exploration","date":"2023-05-24","arxiv_id":"2305.15277","repositories_listed":0,"syntology":null},{"url":"/paper/learnable-behavior-control-breaking-atari","slug":"learnable-behavior-control-breaking-atari","title":"Learnable Behavior Control: Breaking Atari Human World Records via Sample-Efficient Behavior Selection","date":"2023-05-09","arxiv_id":"2305.05239","repositories_listed":0,"syntology":null},{"url":null,"slug":"unlocking-the-power-of-representations-in","title":"Unlocking the Power of Representations in Long-term Novelty-based Exploration","date":"2023-05-02","arxiv_id":"2305.01521","repositories_listed":0,"syntology":null},{"url":null,"slug":"approximate-shielding-of-atari-agents-for","title":"Approximate Shielding of Atari Agents for Safe Exploration","date":"2023-04-21","arxiv_id":"2304.11104","repositories_listed":0,"syntology":null},{"url":null,"slug":"loss-of-plasticity-in-continual-deep","title":"Loss of Plasticity in Continual Deep Reinforcement Learning","date":"2023-03-13","arxiv_id":"2303.07507","repositories_listed":0,"syntology":null},{"url":null,"slug":"double-a3c-deep-reinforcement-learning-on","title":"Double A3C: Deep Reinforcement Learning on OpenAI Gym Games","date":"2023-03-04","arxiv_id":"2303.02271","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-plasticity-in-neural-networks","title":"Understanding plasticity in neural networks","date":"2023-03-02","arxiv_id":"2303.01486","repositories_listed":0,"syntology":null},{"url":null,"slug":"read-and-reap-the-rewards-learning-to-play","title":"Read and Reap the Rewards: Learning to Play Atari with the Help of Instruction Manuals","date":"2023-02-09","arxiv_id":"2302.04449","repositories_listed":0,"syntology":null},{"url":null,"slug":"enabling-surrogate-assisted-evolutionary","title":"Enabling surrogate-assisted evolutionary reinforcement learning via policy embedding","date":"2023-01-31","arxiv_id":"2301.13374","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalized-munchausen-reinforcement-learning","title":"Generalized Munchausen Reinforcement Learning using Tsallis KL Divergence","date":"2023-01-27","arxiv_id":"2301.11476","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-compartment-neuron-and-population","title":"Multi-compartment Neuron and Population Encoding Powered Spiking Neural Network for Deep Distributional Reinforcement Learning","date":"2023-01-18","arxiv_id":"2301.07275","repositories_listed":0,"syntology":null},{"url":null,"slug":"local-guided-global-paired-similarity","title":"Local-Guided Global: Paired Similarity Representation for Visual Reinforcement Learning","date":"2023-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"simultaneously-updating-all-persistence","title":"Simultaneously Updating All Persistence Values in Reinforcement Learning","date":"2022-11-21","arxiv_id":"2211.11620","repositories_listed":0,"syntology":null},{"url":null,"slug":"curiosity-in-hindsight","title":"Curiosity in Hindsight: Intrinsic Exploration in Stochastic Environments","date":"2022-11-18","arxiv_id":"2211.10515","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-rl-with-realistic-datasets","title":"Offline RL With Realistic Datasets: Heteroskedasticity and Support Constraints","date":"2022-11-02","arxiv_id":"2211.01052","repositories_listed":0,"syntology":null},{"url":null,"slug":"spending-thinking-time-wisely-accelerating-1","title":"Spending Thinking Time Wisely: Accelerating MCTS with Virtual Expansions","date":"2022-10-23","arxiv_id":"2210.12628","repositories_listed":0,"syntology":null},{"url":null,"slug":"bayesian-q-learning-with-imperfect-expert","title":"Bayesian Q-learning With Imperfect Expert Demonstrations","date":"2022-10-01","arxiv_id":"2210.01800","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-q-network-for-ai-soccer","title":"Deep Q-Network for AI Soccer","date":"2022-09-20","arxiv_id":"2209.09491","repositories_listed":0,"syntology":null},{"url":null,"slug":"rewarding-episodic-visitation-discrepancy-for","title":"Rewarding Episodic Visitation Discrepancy for Exploration in Reinforcement Learning","date":"2022-09-19","arxiv_id":"2209.08842","repositories_listed":0,"syntology":null},{"url":null,"slug":"emergence-of-novelty-in-evolutionary","title":"Emergence of Novelty in Evolutionary Algorithms","date":"2022-06-27","arxiv_id":"2207.04857","repositories_listed":0,"syntology":null},{"url":"/paper/generalized-data-distribution-iteration","slug":"generalized-data-distribution-iteration","title":"Generalized Data Distribution Iteration","date":"2022-06-07","arxiv_id":"2206.03192","repositories_listed":0,"syntology":null},{"url":null,"slug":"nuclear-norm-maximization-based-curiosity","title":"Nuclear Norm Maximization Based Curiosity-Driven Learning","date":"2022-05-21","arxiv_id":"2205.10484","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-apprenticeship-learning-for-playing","title":"Deep Apprenticeship Learning for Playing Games","date":"2022-05-16","arxiv_id":"2205.07959","repositories_listed":0,"syntology":null},{"url":null,"slug":"memory-constrained-policy-optimization-1","title":"Learning to Constrain Policy Optimization with Virtual Trust Region","date":"2022-04-20","arxiv_id":"2204.09315","repositories_listed":0,"syntology":null},{"url":null,"slug":"methodical-advice-collection-and-reuse-in","title":"Methodical Advice Collection and Reuse in Deep Reinforcement Learning","date":"2022-04-14","arxiv_id":"2204.07254","repositories_listed":0,"syntology":null},{"url":null,"slug":"when-should-we-prefer-offline-reinforcement","title":"When Should We Prefer Offline Reinforcement Learning Over Behavioral Cloning?","date":"2022-04-12","arxiv_id":"2204.05618","repositories_listed":0,"syntology":null},{"url":null,"slug":"mask-atari-for-deep-reinforcement-learning-as","title":"Mask Atari for Deep Reinforcement Learning as POMDP Benchmarks","date":"2022-03-31","arxiv_id":"2203.16777","repositories_listed":0,"syntology":null},{"url":null,"slug":"lazy-mdps-towards-interpretable-reinforcement","title":"Lazy-MDPs: Towards Interpretable Reinforcement Learning by Learning When to Act","date":"2022-03-16","arxiv_id":"2203.08542","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-the-diversity-of-bootstrapped-dqn","title":"Improving the Diversity of Bootstrapped DQN by Replacing Priors With Noise","date":"2022-03-02","arxiv_id":"2203.01004","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-perspective-on-value-backup-and","title":"A Unified Perspective on Value Backup and Exploration in Monte-Carlo Tree Search","date":"2022-02-11","arxiv_id":"2202.07071","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-spiking-q","title":"Deep Reinforcement Learning with Spiking Q-learning","date":"2022-01-21","arxiv_id":"2201.09754","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploration-by-random-network-distillation-1","title":"Exploration by Random Network Distillation","date":"2022-01-17","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-option-critic-1","title":"Attention Option-Critic","date":"2022-01-07","arxiv_id":"2201.02628","repositories_listed":0,"syntology":null},{"url":null,"slug":"execute-order-66-targeted-data-poisoning-for","title":"Execute Order 66: Targeted Data Poisoning for Reinforcement Learning","date":"2022-01-03","arxiv_id":"2201.00762","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-policies-learn","title":"Deep Reinforcement Learning Policies Learn Shared Adversarial Features Across MDPs","date":"2021-12-16","arxiv_id":"2112.09025","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-benchmark-for-low-switching-cost","title":"A Benchmark for Low-Switching-Cost Reinforcement Learning","date":"2021-12-13","arxiv_id":"2112.06424","repositories_listed":0,"syntology":null},{"url":null,"slug":"dr3-value-based-deep-reinforcement-learning-1","title":"DR3: Value-Based Deep Reinforcement Learning Requires Explicit Regularization","date":"2021-12-09","arxiv_id":"2112.04716","repositories_listed":0,"syntology":null}],"record_sha256":"f7e599126f11042389757554a6d8014dc74720a9956d155dc8fae697ca1b9f93","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}