{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/dqn/papers/4","list_of":"/method/dqn","method":"DQN","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":4,"pages_in_order":6,"rows_per_page":100,"rows":[301,400],"of":519,"counts":{"archive_papers_tagged":519,"with_a_code_link":173,"where_syntology_ran_a_sample":47,"not_listed_spam_title":0,"listed":519,"listed_where_code_ran":47,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":36,"every_run_a_failure_of_syntologys_instrument":11,"listed_with_a_run_with_no_instrument_failure":36,"listed_every_run_a_failure_of_syntologys_instrument":11,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/dqn","prev":"/method/dqn/papers/3","next":"/method/dqn/papers/5","papers":[{"paper":"/paper/deep-reinforcement-agent-for-scheduling-in","slug":"deep-reinforcement-agent-for-scheduling-in","title":"Deep Reinforcement Agent for Scheduling in HPC","date":"2021-02-11","arxiv_id":"2102.06243","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-survey-of-motion-planning-algorithms-for","title":"A review of motion planning algorithms for intelligent robotics","date":"2021-02-04","arxiv_id":"2102.02376","n_code_links":0,"syntology":null},{"paper":"/paper/benchmarking-perturbation-based-saliency-maps","slug":"benchmarking-perturbation-based-saliency-maps","title":"Benchmarking Perturbation-based Saliency Maps for Explaining Atari Agents","date":"2021-01-18","arxiv_id":"2101.07312","n_code_links":1,"syntology":null},{"paper":"/paper/action-priors-for-large-action-spaces-in","slug":"action-priors-for-large-action-spaces-in","title":"Action Priors for Large Action Spaces in Robotics","date":"2021-01-11","arxiv_id":"2101.04178","n_code_links":1,"syntology":null},{"paper":"/paper/evolving-reinforcement-learning-algorithms-1","slug":"evolving-reinforcement-learning-algorithms-1","title":"Evolving Reinforcement Learning Algorithms","date":"2021-01-08","arxiv_id":"2101.03958","n_code_links":5,"syntology":null},{"paper":null,"slug":"reinforced-imitative-graph-representation","title":"Reinforced Imitative Graph Representation Learning for Mobile User Profiling: An Adversarial Training Perspective","date":"2021-01-07","arxiv_id":"2101.02634","n_code_links":0,"syntology":null},{"paper":"/paper/reinforcement-learning-with-latent-flow-1","slug":"reinforcement-learning-with-latent-flow-1","title":"Reinforcement Learning with Latent Flow","date":"2021-01-06","arxiv_id":"2101.01857","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["WendyShang/flare","WendyShang/dqn_zoo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/a-novel-policy-for-pre-trained-deep","slug":"a-novel-policy-for-pre-trained-deep","title":"A novel policy for pre-trained Deep Reinforcement Learning for Speech Emotion Recognition","date":"2021-01-04","arxiv_id":"2101.00738","n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-to-search-for-fast-maximum-common","title":"Learning to Search for Fast Maximum Common Subgraph Detection","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"pac-bayesian-randomized-value-function-with","title":"PAC-Bayesian Randomized Value Function with Informative Prior","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"preventing-value-function-collapse-in","title":"Preventing Value Function Collapse in Ensemble Q-Learning by Maximizing Representation Diversity","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"weighted-bellman-backups-for-improved-signal","title":"Weighted Bellman Backups for Improved Signal-to-Noise in Q-Updates","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"disentangled-planning-and-control-in-vision","title":"Disentangled Planning and Control in Vision Based Robotics via Reward Machines","date":"2020-12-28","arxiv_id":"2012.14464","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-state-representation-dueling-network-for","title":"A State Representation Dueling Network for Deep Reinforcement Learning","date":"2020-12-24","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"deploying-reinforcement-learning-in-water","title":"Deploying Reinforcement Learning in Water Transport","date":"2020-12-14","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"learn-to-play-tetris-with-deep-reinforcement","title":"Learn to Play Tetris with Deep Reinforcement Learning","date":"2020-12-14","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"mobile-robots-autonomous-exploration-with","title":"Mobile Robots Autonomous Exploration with Reinforcement Learning","date":"2020-12-14","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/policy-gradient-rl-algorithms-as-directed","slug":"policy-gradient-rl-algorithms-as-directed","title":"Policy Gradient RL Algorithms as Directed Acyclic Graphs","date":"2020-12-14","arxiv_id":"2012.07763","n_code_links":1,"syntology":null},{"paper":null,"slug":"self-correcting-q-learning","title":"Self-correcting Q-Learning","date":"2020-12-02","arxiv_id":"2012.01100","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-new-convergent-variant-of-q-learning-with","title":"A new convergent variant of Q-learning with linear function approximation","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"predictive-per-balancing-priority-and","title":"Predictive PER: Balancing Priority and Diversity towards Stable Deep Reinforcement Learning","date":"2020-11-26","arxiv_id":"2011.13093","n_code_links":0,"syntology":null},{"paper":"/paper/revisiting-rainbow-promoting-more-insightful","slug":"revisiting-rainbow-promoting-more-insightful","title":"Revisiting Rainbow: Promoting more Insightful and Inclusive Deep Reinforcement Learning Research","date":"2020-11-20","arxiv_id":"2011.14826","n_code_links":2,"syntology":null},{"paper":null,"slug":"leveraging-the-variance-of-return-sequences-1","title":"Leveraging the Variance of Return Sequences for Exploration Policy","date":"2020-11-17","arxiv_id":"2011.08649","n_code_links":0,"syntology":null},{"paper":"/paper/optimizing-large-scale-fleet-management-on-a","slug":"optimizing-large-scale-fleet-management-on-a","title":"Optimizing Large-Scale Fleet Management on a Road Network using Multi-Agent Deep Reinforcement Learning with Graph Neural Network","date":"2020-11-12","arxiv_id":"2011.06175","n_code_links":1,"syntology":null},{"paper":"/paper/hamilton-jacobi-deep-q-learning-for","slug":"hamilton-jacobi-deep-q-learning-for","title":"Hamilton-Jacobi Deep Q-Learning for Deterministic Continuous-Time Systems with Lipschitz Continuous Controls","date":"2020-10-27","arxiv_id":"2010.14087","n_code_links":1,"syntology":null},{"paper":null,"slug":"adversarial-attacks-on-deep-algorithmic","title":"Adversarial Attacks on Deep Algorithmic Trading Policies","date":"2020-10-22","arxiv_id":"2010.11388","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-graph-based-and-patient-demographics-aware","title":"A Weighted Heterogeneous Graph Based Dialogue System","date":"2020-10-21","arxiv_id":"2010.10699","n_code_links":0,"syntology":null},{"paper":null,"slug":"chance-constrained-control-with-lexicographic","title":"Chance-Constrained Control with Lexicographic Deep Reinforcement Learning","date":"2020-10-19","arxiv_id":"2010.09468","n_code_links":0,"syntology":null},{"paper":"/paper/connections-between-relational-event-model","slug":"connections-between-relational-event-model","title":"Connections between Relational Event Model and Inverse Reinforcement Learning for Characterizing Group Interaction Sequences","date":"2020-10-19","arxiv_id":"2010.09810","n_code_links":1,"syntology":null},{"paper":null,"slug":"multi-agent-reinforcement-learning-in-noma","title":"Multi-Agent Reinforcement Learning in NOMA-aided UAV Networks for Cellular Offloading","date":"2020-10-18","arxiv_id":"2010.09094","n_code_links":0,"syntology":null},{"paper":null,"slug":"noma-in-uav-aided-cellular-offloading-a","title":"NOMA in UAV-aided cellular offloading: A machine learning approach","date":"2020-10-18","arxiv_id":"2011.14776","n_code_links":0,"syntology":null},{"paper":null,"slug":"machine-learning-empowered-trajectory-and","title":"Machine Learning Empowered Trajectory and Passive Beamforming Design in UAV-RIS Wireless Networks","date":"2020-10-06","arxiv_id":"2010.02749","n_code_links":0,"syntology":null},{"paper":null,"slug":"value-based-bayesian-meta-reinforcement","title":"Bayesian Meta-reinforcement Learning for Traffic Signal Control","date":"2020-10-01","arxiv_id":"2010.00163","n_code_links":0,"syntology":null},{"paper":null,"slug":"strategy-and-benchmark-for-converting-deep-q","title":"Strategy and Benchmark for Converting Deep Q-Networks to Event-Driven Spiking Neural Networks","date":"2020-09-30","arxiv_id":"2009.14456","n_code_links":0,"syntology":null},{"paper":null,"slug":"lineage-evolution-reinforcement-learning","title":"Lineage Evolution Reinforcement Learning","date":"2020-09-26","arxiv_id":"2010.14616","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-new-approach-for-tactical-decision-making","title":"A New Approach for Tactical Decision Making in Lane Changing: Sample Efficient Deep Q Learning with a Safety Feedback Reward","date":"2020-09-24","arxiv_id":"2009.11905","n_code_links":0,"syntology":null},{"paper":null,"slug":"tactical-decision-making-for-emergency","title":"Tactical Decision Making for Emergency Vehicles Based on A Combinational Learning Method","date":"2020-09-09","arxiv_id":"2009.04203","n_code_links":0,"syntology":null},{"paper":"/paper/drle-decentralized-reinforcement-learning-at","slug":"drle-decentralized-reinforcement-learning-at","title":"DRLE: Decentralized Reinforcement Learning at the Edge for Traffic Light Control in the IoV","date":"2020-09-03","arxiv_id":"2009.01502","n_code_links":1,"syntology":null},{"paper":null,"slug":"an-adaptive-synchronization-approach-for","title":"An adaptive synchronization approach for weights of deep reinforcement learning","date":"2020-08-16","arxiv_id":"2008.06973","n_code_links":0,"syntology":null},{"paper":null,"slug":"chrome-dino-run-using-reinforcement-learning","title":"Chrome Dino Run using Reinforcement Learning","date":"2020-08-15","arxiv_id":"2008.06799","n_code_links":0,"syntology":null},{"paper":"/paper/reinforcement-learning-with-quantum","slug":"reinforcement-learning-with-quantum","title":"Reinforcement Learning with Quantum Variational Circuits","date":"2020-08-15","arxiv_id":"2008.07524","n_code_links":2,"syntology":null},{"paper":null,"slug":"generalized-radio-environment-monitoring-for","title":"Generalized Radio Environment Monitoring for Next Generation Wireless Networks","date":"2020-08-14","arxiv_id":"2008.06203","n_code_links":0,"syntology":null},{"paper":null,"slug":"convex-q-learning-part-1-deterministic","title":"Convex Q-Learning, Part 1: Deterministic Optimal Control","date":"2020-08-08","arxiv_id":"2008.03559","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-q-network-based-multi-agent","title":"Deep Q-Network Based Multi-agent Reinforcement Learning with Binary Action Agents","date":"2020-08-06","arxiv_id":"2008.04109","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-for-dynamic-2","title":"Deep Reinforcement Learning for Dynamic Spectrum Sensing and Aggregation in Multi-Channel Wireless Networks","date":"2020-07-28","arxiv_id":"2007.13965","n_code_links":0,"syntology":null},{"paper":null,"slug":"mixture-of-step-returns-in-bootstrapped-dqn","title":"Mixture of Step Returns in Bootstrapped DQN","date":"2020-07-16","arxiv_id":"2007.08229","n_code_links":0,"syntology":null},{"paper":null,"slug":"analysis-of-q-learning-with-adaptation-and","title":"Analysis of Q-learning with Adaptation and Momentum Restart for Gradient Descent","date":"2020-07-15","arxiv_id":"2007.07422","n_code_links":0,"syntology":null},{"paper":null,"slug":"simulating-multi-exit-evacuation-using-deep","title":"Simulating multi-exit evacuation using deep reinforcement learning","date":"2020-07-11","arxiv_id":"2007.05783","n_code_links":0,"syntology":null},{"paper":"/paper/sunrise-a-simple-unified-framework-for","slug":"sunrise-a-simple-unified-framework-for","title":"SUNRISE: A Simple Unified Framework for Ensemble Learning in Deep Reinforcement Learning","date":"2020-07-09","arxiv_id":"2007.04938","n_code_links":1,"syntology":null},{"paper":null,"slug":"auto-map-a-dqn-framework-for-exploring","title":"Auto-MAP: A DQN Framework for Exploring Distributed Execution Plans for DNN Workloads","date":"2020-07-08","arxiv_id":"2007.04069","n_code_links":0,"syntology":null},{"paper":null,"slug":"cognitive-radio-network-throughput","title":"Cognitive Radio Network Throughput Maximization with Deep Reinforcement Learning","date":"2020-07-07","arxiv_id":"2007.03165","n_code_links":0,"syntology":null},{"paper":null,"slug":"decentralized-deep-reinforcement-learning-for-1","title":"Decentralized Deep Reinforcement Learning for Network Level Traffic Signal Control","date":"2020-07-02","arxiv_id":"2007.03433","n_code_links":0,"syntology":null},{"paper":null,"slug":"noise-overestimation-and-exploration-in-deep","title":"Some approaches used to overcome overestimation in Deep Reinforcement Learning algorithms","date":"2020-06-25","arxiv_id":"2006.14167","n_code_links":0,"syntology":null},{"paper":null,"slug":"reducing-overestimation-bias-by-increasing","title":"Preventing Value Function Collapse in Ensemble {Q}-Learning by Maximizing Representation Diversity","date":"2020-06-24","arxiv_id":"2006.13823","n_code_links":0,"syntology":null},{"paper":"/paper/rl-unplugged-benchmarks-for-offline","slug":"rl-unplugged-benchmarks-for-offline","title":"RL Unplugged: A Suite of Benchmarks for Offline Reinforcement Learning","date":"2020-06-24","arxiv_id":"2006.13888","n_code_links":2,"syntology":null},{"paper":"/paper/efficient-ridesharing-dispatch-using-multi","slug":"efficient-ridesharing-dispatch-using-multi","title":"Efficient Ridesharing Dispatch Using Multi-Agent Reinforcement Learning","date":"2020-06-18","arxiv_id":"2006.10897","n_code_links":1,"syntology":null},{"paper":null,"slug":"interaction-networks-using-a-reinforcement","title":"Interaction Networks: Using a Reinforcement Learner to train other Machine Learning algorithms","date":"2020-06-15","arxiv_id":"2006.08457","n_code_links":0,"syntology":null},{"paper":null,"slug":"balancing-a-cartpole-system-with","title":"Balancing a CartPole System with Reinforcement Learning -- A Tutorial","date":"2020-06-08","arxiv_id":"2006.04938","n_code_links":0,"syntology":null},{"paper":"/paper/conservative-q-learning-for-offline","slug":"conservative-q-learning-for-offline","title":"Conservative Q-Learning for Offline Reinforcement Learning","date":"2020-06-08","arxiv_id":"2006.04779","n_code_links":18,"syntology":{"ran":31,"of":34,"n_ran_checked":28,"n_instrument":3,"unverified":3,"pointer_only":19,"phrase":"31 ran (of which 5 constructed an object rather than computing a result; 28 with no instrument failure: 2 honoured, 0 violated, 26 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","official":{"repos":["aviralkumar2907/CQL"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/acme-a-research-framework-for-distributed","slug":"acme-a-research-framework-for-distributed","title":"Acme: A Research Framework for Distributed Reinforcement Learning","date":"2020-06-01","arxiv_id":"2006.00979","n_code_links":5,"syntology":null},{"paper":null,"slug":"learning-to-charge-rf-energy-harvesting","title":"Learning to Charge RF-Energy Harvesting Devices in WiFi Networks","date":"2020-05-25","arxiv_id":"2005.12022","n_code_links":0,"syntology":null},{"paper":null,"slug":"prototypical-q-networks-for-automatic","title":"Prototypical Q Networks for Automatic Conversational Diagnosis and Few-Shot New Disease Adaption","date":"2020-05-19","arxiv_id":"2005.11153","n_code_links":0,"syntology":null},{"paper":"/paper/local-and-global-explanations-of-agent","slug":"local-and-global-explanations-of-agent","title":"Local and Global Explanations of Agent Behavior: Integrating Strategy Summaries with Saliency Maps","date":"2020-05-18","arxiv_id":"2005.08874","n_code_links":1,"syntology":{"ran":9,"of":15,"n_ran_checked":9,"n_instrument":0,"unverified":6,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","official":{"repos":["HuTobias/HIGHLIGHTS-LRP"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"risk-aware-high-level-decisions-for-automated","title":"Risk-Aware High-level Decisions for Automated Driving at Occluded Intersections with Reinforcement Learning","date":"2020-04-09","arxiv_id":"2004.04450","n_code_links":0,"syntology":null},{"paper":"/paper/an-application-of-deep-reinforcement-learning","slug":"an-application-of-deep-reinforcement-learning","title":"An Application of Deep Reinforcement Learning to Algorithmic Trading","date":"2020-04-07","arxiv_id":"2004.06627","n_code_links":1,"syntology":null},{"paper":null,"slug":"uniform-state-abstraction-for-reinforcement","title":"Uniform State Abstraction For Reinforcement Learning","date":"2020-04-06","arxiv_id":"2004.02919","n_code_links":0,"syntology":null},{"paper":null,"slug":"do-recent-advancements-in-model-based-deep-1","title":"Importance of using appropriate baselines for evaluation of data-efficiency in deep reinforcement learning for Atari","date":"2020-03-23","arxiv_id":"2003.10181","n_code_links":0,"syntology":null},{"paper":null,"slug":"interpretable-multi-time-scale-constraints-in","title":"Deep Constrained Q-learning","date":"2020-03-20","arxiv_id":"2003.09398","n_code_links":0,"syntology":null},{"paper":"/paper/robust-deep-reinforcement-learning-against","slug":"robust-deep-reinforcement-learning-against","title":"Robust Deep Reinforcement Learning against Adversarial Perturbations on State Observations","date":"2020-03-19","arxiv_id":"2003.08938","n_code_links":4,"syntology":null},{"paper":"/paper/simultaneous-navigation-and-radio-mapping-for","slug":"simultaneous-navigation-and-radio-mapping-for","title":"Simultaneous Navigation and Radio Mapping for Cellular-Connected UAV with Deep Reinforcement Learning","date":"2020-03-17","arxiv_id":"2003.07574","n_code_links":1,"syntology":null},{"paper":null,"slug":"application-of-deep-q-network-in-portfolio","title":"Application of Deep Q-Network in Portfolio Management","date":"2020-03-13","arxiv_id":"2003.06365","n_code_links":0,"syntology":null},{"paper":null,"slug":"dynamic-experience-replay","title":"Dynamic Experience Replay","date":"2020-03-04","arxiv_id":"2003.02372","n_code_links":0,"syntology":null},{"paper":"/paper/contention-window-optimization-in-ieee","slug":"contention-window-optimization-in-ieee","title":"Contention Window Optimization in IEEE 802.11ax Networks with Deep Reinforcement Learning","date":"2020-03-03","arxiv_id":"2003.01492","n_code_links":1,"syntology":null},{"paper":"/paper/optimistic-exploration-even-with-a-1","slug":"optimistic-exploration-even-with-a-1","title":"Optimistic Exploration even with a Pessimistic Initialisation","date":"2020-02-26","arxiv_id":"2002.12174","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["oxwhirl/opiq"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"disentangling-controllable-object-through","title":"Disentangling Controllable Object through Video Prediction Improves Visual Reinforcement Learning","date":"2020-02-21","arxiv_id":"2002.09136","n_code_links":0,"syntology":null},{"paper":"/paper/langevin-dqn","slug":"langevin-dqn","title":"Langevin DQN","date":"2020-02-17","arxiv_id":"2002.07282","n_code_links":2,"syntology":null},{"paper":null,"slug":"a-multimodal-dialogue-system-for","title":"A Multimodal Dialogue System for Conversational Image Editing","date":"2020-02-16","arxiv_id":"2002.06484","n_code_links":0,"syntology":null},{"paper":"/paper/reinforced-active-learning-for-image-1","slug":"reinforced-active-learning-for-image-1","title":"Reinforced active learning for image segmentation","date":"2020-02-16","arxiv_id":"2002.06583","n_code_links":1,"syntology":null},{"paper":null,"slug":"fast-reinforcement-learning-for-anti-jamming","title":"Fast Reinforcement Learning for Anti-jamming Communications","date":"2020-02-13","arxiv_id":"2002.05364","n_code_links":0,"syntology":null},{"paper":null,"slug":"safe-wasserstein-constrained-deep-q-learning","title":"Safe Wasserstein Constrained Deep Q-Learning","date":"2020-02-07","arxiv_id":"2002.03016","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-rbf-value-functions-for-continuous","title":"Deep Radial-Basis Value Functions for Continuous Control","date":"2020-02-05","arxiv_id":"2002.01883","n_code_links":0,"syntology":null},{"paper":"/paper/interpretable-end-to-end-urban-autonomous","slug":"interpretable-end-to-end-urban-autonomous","title":"Interpretable End-to-end Urban Autonomous Driving with Latent Deep Reinforcement Learning","date":"2020-01-23","arxiv_id":"2001.08726","n_code_links":4,"syntology":{"ran":2,"of":6,"n_ran_checked":2,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["cjy1992/interp-e2e-driving"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"deep-interactive-reinforcement-learning-for","title":"Deep Interactive Reinforcement Learning for Path Following of Autonomous Underwater Vehicle","date":"2020-01-10","arxiv_id":"2001.03359","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-randomized-least-squares-value-iteration","title":"Deep Randomized Least Squares Value Iteration","date":"2020-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/slm-lab-a-comprehensive-benchmark-and-modular-1","slug":"slm-lab-a-comprehensive-benchmark-and-modular-1","title":"SLM Lab: A Comprehensive Benchmark and Modular Software Framework for Reproducible Deep Reinforcement Learning","date":"2019-12-28","arxiv_id":"1912.12482","n_code_links":1,"syntology":{"ran":10,"of":13,"n_ran_checked":10,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["kengz/SLM-Lab"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"exploiting-the-potential-of-deep","title":"Exploiting the potential of deep reinforcement learning for classification tasks in high-dimensional and unstructured data","date":"2019-12-20","arxiv_id":"1912.09595","n_code_links":0,"syntology":null},{"paper":null,"slug":"soft-q-network","title":"Soft Q Network","date":"2019-12-20","arxiv_id":"1912.10891","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-sparse-representations-incrementally","title":"Learning Sparse Representations Incrementally in Deep Reinforcement Learning","date":"2019-12-09","arxiv_id":"1912.04002","n_code_links":0,"syntology":null},{"paper":"/paper/reconciling-returns-with-experience-replay","slug":"reconciling-returns-with-experience-replay","title":"Reconciling λ-Returns with Experience Replay","date":"2019-12-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"placement-optimization-of-aerial-base","title":"Placement Optimization of Aerial Base Stations with Deep Reinforcement Learning","date":"2019-11-19","arxiv_id":"1911.08111","n_code_links":0,"syntology":null},{"paper":null,"slug":"minimalistic-attacks-how-little-it-takes-to","title":"Minimalistic Attacks: How Little it Takes to Fool a Deep Reinforcement Learning Policy","date":"2019-11-10","arxiv_id":"1911.03849","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-end-to-end-deep-rl-framework-for-task","title":"An End-to-End Deep RL Framework for Task Arrangement in Crowdsourcing Platforms","date":"2019-11-04","arxiv_id":"1911.01030","n_code_links":0,"syntology":null},{"paper":"/paper/task-oriented-language-grounding-for-language","slug":"task-oriented-language-grounding-for-language","title":"Task-Oriented Language Grounding for Language Input with Multiple Sub-Goals of Non-Linear Order","date":"2019-10-27","arxiv_id":"1910.12354","n_code_links":1,"syntology":null},{"paper":null,"slug":"momentum-in-reinforcement-learning","title":"Momentum in Reinforcement Learning","date":"2019-10-21","arxiv_id":"1910.09322","n_code_links":0,"syntology":null},{"paper":null,"slug":"resource-allocation-in-mobility-aware","title":"Resource Allocation in Mobility-Aware Federated Learning Networks: A Deep Reinforcement Learning Approach","date":"2019-10-21","arxiv_id":"1910.09172","n_code_links":0,"syntology":null},{"paper":null,"slug":"reverse-experience-replay","title":"Reverse Experience Replay","date":"2019-10-19","arxiv_id":"1910.08780","n_code_links":0,"syntology":null},{"paper":null,"slug":"knowledge-induced-deep-q-network-for-a-slide","title":"Learning Visual Affordances with Target-Orientated Deep Q-Network to Grasp Objects by Harnessing Environmental Fixtures","date":"2019-10-09","arxiv_id":"1910.03781","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-step-greedy-policies-in-model-free-deep-1","title":"Multi-step Greedy Reinforcement Learning Algorithms","date":"2019-10-07","arxiv_id":"1910.02919","n_code_links":0,"syntology":null},{"paper":"/paper/deep-q-network-for-angry-birds","slug":"deep-q-network-for-angry-birds","title":"Deep Q-Network for Angry Birds","date":"2019-10-04","arxiv_id":"1910.01806","n_code_links":1,"syntology":null},{"paper":null,"slug":"im-sorry-dave-im-afraid-i-cant-do-that-deep-q","title":"I'm sorry Dave, I'm afraid I can't do that, Deep Q-learning from forbidden action","date":"2019-10-04","arxiv_id":"1910.02078","n_code_links":0,"syntology":null}],"record_sha256":"3640892e96b7f59f4775d351f7ec182627ac375b7520976e989ec7a46eb85930","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}