{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/dqn/papers/3","list_of":"/method/dqn","method":"DQN","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":3,"pages_in_order":6,"rows_per_page":100,"rows":[201,300],"of":519,"counts":{"archive_papers_tagged":519,"with_a_code_link":173,"where_syntology_ran_a_sample":47,"not_listed_spam_title":0,"listed":519,"listed_where_code_ran":47,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":36,"every_run_a_failure_of_syntologys_instrument":11,"listed_with_a_run_with_no_instrument_failure":36,"listed_every_run_a_failure_of_syntologys_instrument":11,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/dqn","prev":"/method/dqn/papers/2","next":"/method/dqn/papers/4","papers":[{"paper":null,"slug":"enforcing-kl-regularization-in-general","title":"Enforcing KL Regularization in General Tsallis Entropy Reinforcement Learning via Advantage Learning","date":"2022-05-16","arxiv_id":"2205.07885","n_code_links":0,"syntology":null},{"paper":null,"slug":"qhd-a-brain-inspired-hyperdimensional","title":"Efficient Off-Policy Reinforcement Learning via Brain-Inspired Computing","date":"2022-05-14","arxiv_id":"2205.06978","n_code_links":0,"syntology":null},{"paper":null,"slug":"characterizing-the-action-generalization-gap","title":"Characterizing the Action-Generalization Gap in Deep Q-Learning","date":"2022-05-11","arxiv_id":"2205.05588","n_code_links":0,"syntology":null},{"paper":"/paper/epida-an-easy-plug-in-data-augmentation","slug":"epida-an-easy-plug-in-data-augmentation","title":"EPiDA: An Easy Plug-in Data Augmentation Framework for High Performance Text Classification","date":"2022-04-24","arxiv_id":"2204.11205","n_code_links":1,"syntology":{"ran":2,"of":6,"n_ran_checked":2,"n_instrument":0,"unverified":4,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["zhaominyiz/epida"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"graph-neural-network-based-agent-in-google","title":"Graph Neural Network based Agent in Google Research Football","date":"2022-04-23","arxiv_id":"2204.11142","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-how-to-interact-with-a-complex","title":"Learning how to Interact with a Complex Interface using Hierarchical Reinforcement Learning","date":"2022-04-21","arxiv_id":"2204.10374","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-re-ranking-with-2d-grid-based","title":"Reinforcement Re-ranking with 2D Grid-based Recommendation Panels","date":"2022-04-11","arxiv_id":"2204.04954","n_code_links":0,"syntology":null},{"paper":"/paper/douzero-improving-doudizhu-ai-by-opponent","slug":"douzero-improving-doudizhu-ai-by-opponent","title":"DouZero+: Improving DouDizhu AI by Opponent Modeling and Coach-guided Learning","date":"2022-04-06","arxiv_id":"2204.02558","n_code_links":1,"syntology":null},{"paper":null,"slug":"rem-routing-entropy-minimization-for-capsule","title":"REM: Routing Entropy Minimization for Capsule Networks","date":"2022-04-04","arxiv_id":"2204.01298","n_code_links":0,"syntology":null},{"paper":"/paper/hysteresis-based-rl-robustifying","slug":"hysteresis-based-rl-robustifying","title":"Hysteresis-Based RL: Robustifying Reinforcement Learning-based Control Policies via Hybrid Control","date":"2022-04-01","arxiv_id":"2204.00654","n_code_links":2,"syntology":null},{"paper":null,"slug":"investigating-the-properties-of-neural","title":"Investigating the Properties of Neural Network Representations in Reinforcement Learning","date":"2022-03-30","arxiv_id":"2203.15955","n_code_links":0,"syntology":null},{"paper":null,"slug":"merlin-malware-evasion-with-reinforcement","title":"MERLIN -- Malware Evasion with Reinforcement LearnINg","date":"2022-03-24","arxiv_id":"2203.12980","n_code_links":0,"syntology":null},{"paper":null,"slug":"does-dqn-really-learn-exploring-adversarial","title":"Does DQN really learn? Exploring adversarial training schemes in Pong","date":"2022-03-20","arxiv_id":"2203.10614","n_code_links":0,"syntology":null},{"paper":null,"slug":"random-ensemble-reinforcement-learning-for","title":"Random Ensemble Reinforcement Learning for Traffic Signal Control","date":"2022-03-10","arxiv_id":"2203.05961","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-based-model-free","title":"Deep Reinforcement Learning based Model-free On-line Dynamic Multi-Microgrid Formation to Enhance Resilience","date":"2022-03-06","arxiv_id":"2203.03030","n_code_links":0,"syntology":null},{"paper":null,"slug":"retrieval-augmented-reinforcement-learning-1","title":"Retrieval-Augmented Reinforcement Learning","date":"2022-02-17","arxiv_id":"2202.08417","n_code_links":0,"syntology":null},{"paper":"/paper/skrl-modular-and-flexible-library-for","slug":"skrl-modular-and-flexible-library-for","title":"skrl: Modular and Flexible Library for Reinforcement Learning","date":"2022-02-08","arxiv_id":"2202.03825","n_code_links":1,"syntology":null},{"paper":null,"slug":"generative-adversarial-exploration-for","title":"Generative Adversarial Exploration for Reinforcement Learning","date":"2022-01-27","arxiv_id":"2201.11685","n_code_links":0,"syntology":null},{"paper":null,"slug":"critic-algorithms-using-cooperative-networks","title":"Critic Algorithms using Cooperative Networks","date":"2022-01-19","arxiv_id":"2201.07839","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-improved-reinforcement-learning-algorithm","title":"An Improved Reinforcement Learning Algorithm for Learning to Branch","date":"2022-01-17","arxiv_id":"2201.06213","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-resolution-enhancement-plug-in-for","title":"A Resolution Enhancement Plug-in for Deformable Registration of Medical Images","date":"2021-12-30","arxiv_id":"2112.15180","n_code_links":0,"syntology":null},{"paper":"/paper/constraint-sampling-reinforcement-learning","slug":"constraint-sampling-reinforcement-learning","title":"Constraint Sampling Reinforcement Learning: Incorporating Expertise For Faster Learning","date":"2021-12-30","arxiv_id":"2112.15221","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-graph-attention-learning-approach-to","title":"A Graph Attention Learning Approach to Antenna Tilt Optimization","date":"2021-12-27","arxiv_id":"2112.14843","n_code_links":0,"syntology":null},{"paper":"/paper/intelligent-traffic-light-via-policy-based","slug":"intelligent-traffic-light-via-policy-based","title":"Intelligent Traffic Light via Policy-based Deep Reinforcement Learning","date":"2021-12-27","arxiv_id":"2112.13817","n_code_links":1,"syntology":null},{"paper":"/paper/lane-change-decision-making-through-deep-1","slug":"lane-change-decision-making-through-deep-1","title":"Lane Change Decision-Making through Deep Reinforcement Learning","date":"2021-12-24","arxiv_id":"2112.14705","n_code_links":2,"syntology":null},{"paper":null,"slug":"local-advantage-networks-for-cooperative","title":"Local Advantage Networks for Cooperative Multi-Agent Reinforcement Learning","date":"2021-12-23","arxiv_id":"2112.12458","n_code_links":0,"syntology":null},{"paper":null,"slug":"aerial-base-station-positioning-and-power","title":"Aerial Base Station Positioning and Power Control for Securing Communications: A Deep Q-Network Approach","date":"2021-12-21","arxiv_id":"2112.11090","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-deep-reinforcement-learning-model-for","title":"A deep reinforcement learning model for predictive maintenance planning of road assets: Integrating LCA and LCCA","date":"2021-12-20","arxiv_id":"2112.12589","n_code_links":0,"syntology":null},{"paper":"/paper/space-non-cooperative-object-active-tracking","slug":"space-non-cooperative-object-active-tracking","title":"Space Non-cooperative Object Active Tracking with Deep Reinforcement Learning","date":"2021-12-18","arxiv_id":"2112.09854","n_code_links":1,"syntology":null},{"paper":null,"slug":"scientific-discovery-and-the-cost-of","title":"Scientific Discovery and the Cost of Measurement -- Balancing Information and Cost in Reinforcement Learning","date":"2021-12-14","arxiv_id":"2112.07535","n_code_links":0,"syntology":null},{"paper":"/paper/deep-q-network-with-proximal-iteration-1","slug":"deep-q-network-with-proximal-iteration-1","title":"Faster Deep Reinforcement Learning with Slower Online Network","date":"2021-12-10","arxiv_id":"2112.05848","n_code_links":1,"syntology":{"ran":7,"of":9,"n_ran_checked":7,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["amazon-research/fast-rl-with-slow-updates"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/gdi-rethinking-what-makes-reinforcement-1","slug":"gdi-rethinking-what-makes-reinforcement-1","title":"GDI: Rethinking What Makes Reinforcement Learning Different from Supervised Learning","date":"2021-11-24","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"an-improved-reinforcement-learning-model","title":"An Improved Reinforcement Learning Model Based on Sentiment Analysis","date":"2021-11-19","arxiv_id":"2111.15354","n_code_links":0,"syntology":null},{"paper":null,"slug":"improved-method-of-stock-trading-under","title":"Improved Method of Stock Trading under Reinforcement Learning Based on DRQN and Sentiment Indicators ARBR","date":"2021-11-19","arxiv_id":"2111.15356","n_code_links":0,"syntology":null},{"paper":"/paper/improving-experience-replay-through-modeling","slug":"improving-experience-replay-through-modeling","title":"Improving Experience Replay through Modeling of Similar Transitions' Sets","date":"2021-11-12","arxiv_id":"2111.06907","n_code_links":1,"syntology":null},{"paper":"/paper/good-robot-now-watch-this-repurposing","slug":"good-robot-now-watch-this-repurposing","title":"\"Good Robot! Now Watch This!\": Repurposing Reinforcement Learning for Task-to-Task Transfer","date":"2021-11-08","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/guiding-multi-step-rearrangement-tasks-with","slug":"guiding-multi-step-rearrangement-tasks-with","title":"Guiding Multi-Step Rearrangement Tasks with Natural Language Instructions","date":"2021-11-08","arxiv_id":null,"n_code_links":2,"syntology":null},{"paper":null,"slug":"improving-rna-secondary-structure-design","title":"Improving RNA Secondary Structure Design using Deep Reinforcement Learning","date":"2021-11-05","arxiv_id":"2111.04504","n_code_links":0,"syntology":null},{"paper":null,"slug":"online-service-provisioning-in-nfv-enabled","title":"Online Service Provisioning in NFV-enabled Networks Using Deep Reinforcement Learning","date":"2021-11-03","arxiv_id":"2111.02209","n_code_links":0,"syntology":null},{"paper":"/paper/human-level-control-without-server-grade-1","slug":"human-level-control-without-server-grade-1","title":"Human-Level Control without Server-Grade Hardware","date":"2021-11-01","arxiv_id":"2111.01264","n_code_links":1,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-aided-packet","title":"Deep Reinforcement Learning Aided Packet-Routing For Aeronautical Ad-Hoc Networks Formed by Passenger Planes","date":"2021-10-28","arxiv_id":"2110.15146","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-dpdk-based-acceleration-method-for","title":"Accelerating Distributed Deep Reinforcement Learning by In-Network Experience Sampling","date":"2021-10-26","arxiv_id":"2110.13506","n_code_links":0,"syntology":null},{"paper":"/paper/distributional-reinforcement-learning-for-4","slug":"distributional-reinforcement-learning-for-4","title":"Distributional Reinforcement Learning for Multi-Dimensional Reward Functions","date":"2021-10-26","arxiv_id":"2110.13578","n_code_links":2,"syntology":null},{"paper":null,"slug":"can-q-learning-solve-multi-armed-bantids","title":"Can Q-learning solve Multi Armed Bantids?","date":"2021-10-21","arxiv_id":"2110.10934","n_code_links":0,"syntology":null},{"paper":null,"slug":"more-efficient-exploration-with-symbolic","title":"More Efficient Exploration with Symbolic Priors on Action Sequence Equivalences","date":"2021-10-20","arxiv_id":"2110.10632","n_code_links":0,"syntology":null},{"paper":"/paper/bridging-the-gap-between-label-and-reference-1","slug":"bridging-the-gap-between-label-and-reference-1","title":"Bridging the Gap between Label- and Reference-based Synthesis in Multi-attribute Image-to-Image Translation","date":"2021-10-11","arxiv_id":"2110.05055","n_code_links":1,"syntology":null},{"paper":null,"slug":"navigation-in-urban-environments-amongst","title":"Navigation In Urban Environments Amongst Pedestrians Using Multi-Objective Deep Reinforcement Learning","date":"2021-10-11","arxiv_id":"2110.05205","n_code_links":0,"syntology":null},{"paper":"/paper/reinforcement-learning-in-two-player-zero-sum","slug":"reinforcement-learning-in-two-player-zero-sum","title":"Reinforcement Learning In Two Player Zero Sum Simultaneous Action Games","date":"2021-10-10","arxiv_id":"2110.04835","n_code_links":1,"syntology":null},{"paper":null,"slug":"parallel-actors-and-learners-a-framework-for","title":"Parallel Actors and Learners: A Framework for Generating Scalable RL Implementations","date":"2021-10-03","arxiv_id":"2110.01101","n_code_links":0,"syntology":null},{"paper":null,"slug":"motion-planning-for-autonomous-vehicles-in","title":"Motion Planning for Autonomous Vehicles in the Presence of Uncertainty Using Reinforcement Learning","date":"2021-10-01","arxiv_id":"2110.00640","n_code_links":0,"syntology":null},{"paper":null,"slug":"better-state-exploration-using-action","title":"Better state exploration using action sequence equivalence","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"bootstrapped-hindsight-experience-replay-with","title":"Bootstrapped Hindsight Experience replay with Counterintuitive Prioritization","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"convergent-and-efficient-deep-q-learning","title":"Convergent and Efficient Deep Q Learning Algorithm","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/deep-reinforcement-q-learning-for-intelligent","slug":"deep-reinforcement-q-learning-for-intelligent","title":"Deep Reinforcement Q-Learning for Intelligent Traffic Signal Control with Partial Detection","date":"2021-09-29","arxiv_id":"2109.14337","n_code_links":1,"syntology":null},{"paper":null,"slug":"disentangling-generalization-in-reinforcement","title":"Disentangling Generalization in Reinforcement Learning","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/explanation-aware-experience-replay-in-rule","slug":"explanation-aware-experience-replay-in-rule","title":"Explanation-Aware Experience Replay in Rule-Dense Environments","date":"2021-09-29","arxiv_id":"2109.14711","n_code_links":1,"syntology":null},{"paper":"/paper/hyperdqn-a-randomized-exploration-method-for","slug":"hyperdqn-a-randomized-exploration-method-for","title":"HyperDQN: A Randomized Exploration Method for Deep Reinforcement Learning","date":"2021-09-29","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"the-guide-and-the-explorer-smart-agents-for","title":"The guide and the explorer: smart agents for resource-limited iterated batch reinforcement learning","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"fetal-oxygen-delivery-and-consumption-and","title":"Fetal oxygen delivery and consumption and blood gases in relation to gestational age","date":"2021-09-23","arxiv_id":"2109.11616","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-on-encrypted-data","title":"Reinforcement Learning on Encrypted Data","date":"2021-09-16","arxiv_id":"2109.08236","n_code_links":0,"syntology":null},{"paper":null,"slug":"vision-transformer-for-learning-driving","title":"Vision Transformer for Learning Driving Policies in Complex Multi-Agent Environments","date":"2021-09-14","arxiv_id":"2109.06514","n_code_links":0,"syntology":null},{"paper":"/paper/memory-semantization-through-perturbed-and","slug":"memory-semantization-through-perturbed-and","title":"Learning cortical representations through perturbed and adversarial dreaming","date":"2021-09-09","arxiv_id":"2109.04261","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["NicoZenith/PAD"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/temporal-aware-deep-reinforcement-learning","slug":"temporal-aware-deep-reinforcement-learning","title":"Temporal Shift Reinforcement Learning","date":"2021-09-05","arxiv_id":"2109.02145","n_code_links":1,"syntology":null},{"paper":"/paper/deep-reinforcement-learning-at-the-edge-of","slug":"deep-reinforcement-learning-at-the-edge-of","title":"Deep Reinforcement Learning at the Edge of the Statistical Precipice","date":"2021-08-30","arxiv_id":"2108.13264","n_code_links":3,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 4 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["google-research/rliable"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"a-microscopic-pandemic-simulator-for-pandemic","title":"A Microscopic Pandemic Simulator for Pandemic Prediction Using Scalable Million-Agent Reinforcement Learning","date":"2021-08-14","arxiv_id":"2108.06589","n_code_links":0,"syntology":null},{"paper":"/paper/dqn-control-solution-for-kdd-cup-2021-city","slug":"dqn-control-solution-for-kdd-cup-2021-city","title":"DQN Control Solution for KDD Cup 2021 City Brain Challenge","date":"2021-08-14","arxiv_id":"2108.06491","n_code_links":1,"syntology":null},{"paper":null,"slug":"two-is-a-crowd-tracking-relations-in-videos","title":"Two is a crowd: tracking relations in videos","date":"2021-08-11","arxiv_id":"2108.05331","n_code_links":0,"syntology":null},{"paper":null,"slug":"modified-double-dqn-addressing-stability","title":"Modified Double DQN: addressing stability","date":"2021-08-09","arxiv_id":"2108.04115","n_code_links":0,"syntology":null},{"paper":"/paper/an-efficient-image-to-image-translation","slug":"an-efficient-image-to-image-translation","title":"An Efficient Image-to-Image Translation HourGlass-based Architecture for Object Pushing Policy Learning","date":"2021-08-02","arxiv_id":"2108.01034","n_code_links":1,"syntology":null},{"paper":"/paper/a-dqn-based-approach-to-finding-precise","slug":"a-dqn-based-approach-to-finding-precise","title":"A DQN-based Approach to Finding Precise Evidences for Fact Verification","date":"2021-08-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"rem-efficient-semi-automated-real-time","title":"REM: Efficient Semi-Automated Real-Time Moderation of Online Forums","date":"2021-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"an-improved-algorithm-of-robot-path-planning","title":"An Improved Algorithm of Robot Path Planning in Complex Environment Based on Double DQN","date":"2021-07-23","arxiv_id":"2107.11245","n_code_links":0,"syntology":null},{"paper":"/paper/a-reinforcement-learning-environment-for-2","slug":"a-reinforcement-learning-environment-for-2","title":"A Reinforcement Learning Environment for Mathematical Reasoning via Program Synthesis","date":"2021-07-15","arxiv_id":"2107.07373","n_code_links":1,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-based-dynamic-2","title":"Deep Reinforcement Learning based Dynamic Optimization of Bus Timetable","date":"2021-07-15","arxiv_id":"2107.07066","n_code_links":0,"syntology":null},{"paper":null,"slug":"minimizing-safety-interference-for-safe-and","title":"Minimizing Safety Interference for Safe and Comfortable Automated Driving with Distributional Reinforcement Learning","date":"2021-07-15","arxiv_id":"2107.07316","n_code_links":0,"syntology":null},{"paper":"/paper/improve-agents-without-retraining-parallel","slug":"improve-agents-without-retraining-parallel","title":"Improve Agents without Retraining: Parallel Tree Search with Off-Policy Correction","date":"2021-07-04","arxiv_id":"2107.01715","n_code_links":1,"syntology":null},{"paper":null,"slug":"aoi-minimization-in-energy-harvesting-and","title":"AoI Minimization in Energy Harvesting and Spectrum Sharing Enabled 6G Networks","date":"2021-07-01","arxiv_id":"2107.00340","n_code_links":0,"syntology":null},{"paper":"/paper/a-convergent-and-efficient-deep-q-network","slug":"a-convergent-and-efficient-deep-q-network","title":"Convergent and Efficient Deep Q Network Algorithm","date":"2021-06-29","arxiv_id":"2106.15419","n_code_links":1,"syntology":null},{"paper":null,"slug":"mmd-mix-value-function-factorisation-with","title":"MMD-MIX: Value Function Factorisation with Maximum Mean Discrepancy for Cooperative Multi-Agent Reinforcement Learning","date":"2021-06-22","arxiv_id":"2106.11652","n_code_links":0,"syntology":null},{"paper":"/paper/vision-language-navigation-with-random","slug":"vision-language-navigation-with-random","title":"Vision-Language Navigation with Random Environmental Mixup","date":"2021-06-15","arxiv_id":"2106.07876","n_code_links":1,"syntology":{"ran":1,"of":4,"n_ran_checked":0,"n_instrument":1,"unverified":3,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["lcfractal/vlnrem"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/douzero-mastering-doudizhu-with-self-play","slug":"douzero-mastering-doudizhu-with-self-play","title":"DouZero: Mastering DouDizhu with Self-Play Deep Reinforcement Learning","date":"2021-06-11","arxiv_id":"2106.06135","n_code_links":1,"syntology":{"ran":3,"of":6,"n_ran_checked":2,"n_instrument":1,"unverified":3,"pointer_only":1,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["kwai/DouZero"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["community","official"]}}},{"paper":"/paper/gdi-rethinking-what-makes-reinforcement","slug":"gdi-rethinking-what-makes-reinforcement","title":"GDI: Rethinking What Makes Reinforcement Learning Different From Supervised Learning","date":"2021-06-11","arxiv_id":"2106.06232","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforced-few-shot-acquisition-function","title":"Reinforced Few-Shot Acquisition Function Learning for Bayesian Optimization","date":"2021-06-08","arxiv_id":"2106.04335","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-to-optimize-industry-scale-dynamic","title":"Learning to Optimize Industry-Scale Dynamic Pickup and Delivery Problems","date":"2021-05-27","arxiv_id":"2105.12899","n_code_links":0,"syntology":null},{"paper":"/paper/improved-exploring-starts-by-kernel-density","slug":"improved-exploring-starts-by-kernel-density","title":"Improved Exploring Starts by Kernel Density Estimation-Based State-Space Coverage Acceleration in Reinforcement Learning","date":"2021-05-19","arxiv_id":"2105.08990","n_code_links":1,"syntology":null},{"paper":null,"slug":"reinforcement-learning-with-expert-trajectory","title":"Reinforcement Learning with Expert Trajectory For Quantitative Trading","date":"2021-05-09","arxiv_id":"2105.03844","n_code_links":0,"syntology":null},{"paper":null,"slug":"time-aware-q-networks-resolving-temporal","title":"Time-Aware Q-Networks: Resolving Temporal Irregularity for Deep Reinforcement Learning","date":"2021-05-06","arxiv_id":"2105.02580","n_code_links":0,"syntology":null},{"paper":"/paper/automated-scoring-of-pre-rem-sleep-in-mice","slug":"automated-scoring-of-pre-rem-sleep-in-mice","title":"Automated scoring of pre-REM sleep in mice with deep learning","date":"2021-05-05","arxiv_id":"2105.01933","n_code_links":1,"syntology":null},{"paper":"/paper/adapting-to-reward-progressivity-via-spectral-1","slug":"adapting-to-reward-progressivity-via-spectral-1","title":"Adapting to Reward Progressivity via Spectral Reinforcement Learning","date":"2021-04-29","arxiv_id":"2104.14138","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","official":{"repos":["mchldann/SpectralDQN"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"antagonistic-crowd-simulation-model","title":"Emotional Contagion-Aware Deep Reinforcement Learning for Antagonistic Crowd Simulation","date":"2021-04-29","arxiv_id":"2105.00854","n_code_links":0,"syntology":null},{"paper":"/paper/independent-reinforcement-learning-for-weakly","slug":"independent-reinforcement-learning-for-weakly","title":"Independent Reinforcement Learning for Weakly Cooperative Multiagent Traffic Control Problem","date":"2021-04-22","arxiv_id":"2104.10917","n_code_links":1,"syntology":null},{"paper":"/paper/a-coevolutionairy-approach-to-deep-multi","slug":"a-coevolutionairy-approach-to-deep-multi","title":"A coevolutionary approach to deep multi-agent reinforcement learning","date":"2021-04-12","arxiv_id":"2104.05610","n_code_links":1,"syntology":null},{"paper":null,"slug":"full-gradient-dqn-reinforcement-learning-a","title":"Full Gradient DQN Reinforcement Learning: A Provably Convergent Scheme","date":"2021-03-10","arxiv_id":"2103.05981","n_code_links":0,"syntology":null},{"paper":null,"slug":"increasing-energy-efficiency-of-massive-mimo","title":"Increasing Energy Efficiency of Massive-MIMO Network via Base Stations Switching using Reinforcement Learning and Radio Environment Maps","date":"2021-03-08","arxiv_id":"2103.11891","n_code_links":0,"syntology":null},{"paper":"/paper/super-resolving-compressed-images-via","slug":"super-resolving-compressed-images-via","title":"Super-resolving Compressed Images via Parallel and Series Integration of Artifact Reduction and Resolution Enhancement","date":"2021-03-02","arxiv_id":"2103.01698","n_code_links":1,"syntology":null},{"paper":null,"slug":"multi-agent-path-planning-based-on-mpc-and","title":"Multi-Agent Path Planning based on MPC and DDPG","date":"2021-02-26","arxiv_id":"2102.13283","n_code_links":0,"syntology":null},{"paper":null,"slug":"greedy-multi-step-off-policy-reinforcement-1","title":"Greedy-Step Off-Policy Reinforcement Learning","date":"2021-02-23","arxiv_id":"2102.11717","n_code_links":0,"syntology":null},{"paper":null,"slug":"stratified-experience-replay-correcting","title":"Stratified Experience Replay: Correcting Multiplicity Bias in Off-Policy Reinforcement Learning","date":"2021-02-22","arxiv_id":"2102.11319","n_code_links":0,"syntology":null},{"paper":"/paper/causal-inference-q-network-toward-resilient-1","slug":"causal-inference-q-network-toward-resilient-1","title":"Training a Resilient Q-Network against Observational Interference","date":"2021-02-18","arxiv_id":"2102.09677","n_code_links":1,"syntology":null},{"paper":"/paper/recurrent-rational-networks","slug":"recurrent-rational-networks","title":"Adaptive Rational Activations to Boost Deep Reinforcement Learning","date":"2021-02-18","arxiv_id":"2102.09407","n_code_links":4,"syntology":{"ran":6,"of":6,"n_ran_checked":0,"n_instrument":6,"unverified":0,"pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ml-research/rational_activations","ml-research/rational_rl","ml-research/rational_sl","k4ntz/activation-functions"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"ab7c79f32900712be315d84ded0b789af64e80c93855408006cbf627bc1fce9a","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}