{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/dqn/papers/2","list_of":"/method/dqn","method":"DQN","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":2,"pages_in_order":6,"rows_per_page":100,"rows":[101,200],"of":519,"counts":{"archive_papers_tagged":519,"with_a_code_link":173,"where_syntology_ran_a_sample":47,"not_listed_spam_title":0,"listed":519,"listed_where_code_ran":47,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":36,"every_run_a_failure_of_syntologys_instrument":11,"listed_with_a_run_with_no_instrument_failure":36,"listed_every_run_a_failure_of_syntologys_instrument":11,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/dqn","prev":"/method/dqn","next":"/method/dqn/papers/3","papers":[{"paper":null,"slug":"wake-sleep-consolidated-learning","title":"Wake-Sleep Consolidated Learning","date":"2023-12-06","arxiv_id":"2401.08623","n_code_links":0,"syntology":null},{"paper":null,"slug":"lights-out-training-rl-agents-robust-to","title":"Lights out: training RL agents robust to temporary blindness","date":"2023-12-05","arxiv_id":"2312.02665","n_code_links":0,"syntology":null},{"paper":"/paper/adsorbrl-deep-multi-objective-reinforcement","slug":"adsorbrl-deep-multi-objective-reinforcement","title":"AdsorbRL: Deep Multi-Objective Reinforcement Learning for Inverse Catalysts Design","date":"2023-12-04","arxiv_id":"2312.02308","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["rlacombe/adsorbrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"self-driving-telescopes-autonomous-scheduling","title":"Self-Driving Telescopes: Autonomous Scheduling of Astronomical Observation Campaigns with Offline Reinforcement Learning","date":"2023-11-29","arxiv_id":"2311.18094","n_code_links":0,"syntology":null},{"paper":"/paper/reinforcement-learning-for-wildfire","slug":"reinforcement-learning-for-wildfire","title":"Reinforcement Learning for Wildfire Mitigation in Simulated Disaster Environments","date":"2023-11-27","arxiv_id":"2311.15925","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":5,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["mitrefireline/simfire"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"two-compartment-neuronal-spiking-model","title":"Two-compartment neuronal spiking model expressing brain-state specific apical-amplification, -isolation and -drive regimes","date":"2023-11-10","arxiv_id":"2311.06074","n_code_links":0,"syntology":null},{"paper":null,"slug":"advancing-algorithmic-trading-a-multi","title":"Advancing Algorithmic Trading: A Multi-Technique Enhancement of Deep Q-Network Models","date":"2023-11-09","arxiv_id":"2311.05743","n_code_links":0,"syntology":null},{"paper":"/paper/selectively-sharing-experiences-improves","slug":"selectively-sharing-experiences-improves","title":"Selectively Sharing Experiences Improves Multi-Agent Reinforcement Learning","date":"2023-11-01","arxiv_id":"2311.00865","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["mgerstgrasser/super"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"automaton-distillation-neuro-symbolic","title":"Automaton Distillation: Neuro-Symbolic Transfer Learning for Deep Reinforcement Learning","date":"2023-10-29","arxiv_id":"2310.19137","n_code_links":0,"syntology":null},{"paper":"/paper/weakly-coupled-deep-q-networks","slug":"weakly-coupled-deep-q-networks","title":"Weakly Coupled Deep Q-Networks","date":"2023-10-28","arxiv_id":"2310.18803","n_code_links":0,"syntology":{"ran":5,"of":10,"n_ran_checked":3,"n_instrument":2,"unverified":5,"pointer_only":10,"phrase":"5 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","official":null}},{"paper":null,"slug":"on-the-convergence-and-sample-complexity","title":"On the Convergence and Sample Complexity Analysis of Deep Q-Networks with $ε$-Greedy Exploration","date":"2023-10-24","arxiv_id":"2310.16173","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-based-local-path","title":"Reinforcement learning based local path planning for mobile robot","date":"2023-10-24","arxiv_id":"2403.12463","n_code_links":0,"syntology":null},{"paper":"/paper/deep-reinforcement-learning-based-intelligent-2","slug":"deep-reinforcement-learning-based-intelligent-2","title":"Deep Reinforcement Learning-based Intelligent Traffic Signal Controls with Optimized CO2 emissions","date":"2023-10-19","arxiv_id":"2310.13129","n_code_links":1,"syntology":null},{"paper":null,"slug":"automatic-music-playlist-generation-via","title":"Automatic Music Playlist Generation via Simulation-based Reinforcement Learning","date":"2023-10-13","arxiv_id":"2310.09123","n_code_links":0,"syntology":null},{"paper":null,"slug":"optimal-scheduling-of-electric-vehicle","title":"Optimal Scheduling of Electric Vehicle Charging with Deep Reinforcement Learning considering End Users Flexibility","date":"2023-10-13","arxiv_id":"2310.09040","n_code_links":0,"syntology":null},{"paper":"/paper/learning-rl-policies-for-joint-beamforming","slug":"learning-rl-policies-for-joint-beamforming","title":"Learning RL-Policies for Joint Beamforming Without Exploration: A Batch Constrained Off-Policy Approach","date":"2023-10-12","arxiv_id":"2310.08660","n_code_links":1,"syntology":null},{"paper":null,"slug":"optimal-sequential-decision-making-in","title":"Optimal Sequential Decision-Making in Geosteering: A Reinforcement Learning Approach","date":"2023-10-07","arxiv_id":"2310.04772","n_code_links":0,"syntology":null},{"paper":"/paper/pgdqn-preference-guided-deep-q-network","slug":"pgdqn-preference-guided-deep-q-network","title":"PGDQN: Preference-Guided Deep Q-Network","date":"2023-10-03","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"universal-sleep-decoder-aligning-awake-and","title":"SI-SD: Sleep Interpreter through awake-guided cross-subject Semantic Decoding","date":"2023-09-28","arxiv_id":"2309.16457","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-for-the-heat","title":"Deep Reinforcement Learning for the Heat Transfer Control of Pulsating Impinging Jets","date":"2023-09-25","arxiv_id":"2309.13955","n_code_links":0,"syntology":null},{"paper":"/paper/enhancing-data-efficiency-in-reinforcement","slug":"enhancing-data-efficiency-in-reinforcement","title":"Enhancing data efficiency in reinforcement learning: a novel imagination mechanism based on mesh information propagation","date":"2023-09-25","arxiv_id":"2309.14243","n_code_links":2,"syntology":null},{"paper":null,"slug":"enhancing-healthcare-with-eog-a-novel","title":"Enhancing Healthcare with EOG: A Novel Approach to Sleep Stage Classification","date":"2023-09-25","arxiv_id":"2310.03757","n_code_links":0,"syntology":null},{"paper":null,"slug":"implicit-sensing-in-traffic-optimization","title":"Implicit Sensing in Traffic Optimization: Advanced Deep Reinforcement Learning Techniques","date":"2023-09-25","arxiv_id":"2309.14395","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-convergence-and-sample-complexity-1","title":"On the Convergence and Sample Complexity Analysis of Deep Q-Networks with $\\epsilon$-Greedy Exploration","date":"2023-09-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"ai-driven-patient-monitoring-with-multi-agent","title":"Adaptive Multi-Agent Deep Reinforcement Learning for Timely Healthcare Interventions","date":"2023-09-20","arxiv_id":"2309.10980","n_code_links":0,"syntology":null},{"paper":null,"slug":"coalescent-processes-emerging-from-large","title":"Coalescent processes emerging from large deviations","date":"2023-08-28","arxiv_id":"2308.14715","n_code_links":0,"syntology":null},{"paper":"/paper/llm-powered-sim-to-real-transfer-for-traffic","slug":"llm-powered-sim-to-real-transfer-for-traffic","title":"Prompt to Transfer: Sim-to-Real Transfer for Traffic Signal Control with Prompt Learning","date":"2023-08-28","arxiv_id":"2308.14284","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["darl-libsignal/promptgat"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"understanding-the-usage-of-qubo-based","title":"A Graph Neural Network-Based QUBO-Formulated Hamiltonian-Inspired Loss Function for Combinatorial Optimization using Reinforcement Learning","date":"2023-08-27","arxiv_id":"2308.13978","n_code_links":0,"syntology":null},{"paper":null,"slug":"integrating-renewable-energy-in-agriculture-a","title":"Integrating Renewable Energy in Agriculture: A Deep Reinforcement Learning-based Approach","date":"2023-08-16","arxiv_id":"2308.08611","n_code_links":0,"syntology":null},{"paper":null,"slug":"bag-of-policies-for-distributional-deep","title":"Bag of Policies for Distributional Deep Exploration","date":"2023-08-03","arxiv_id":"2308.01759","n_code_links":0,"syntology":null},{"paper":null,"slug":"caching-at-stars-the-next-generation-edge","title":"Caching-at-STARS: the Next Generation Edge Caching","date":"2023-08-01","arxiv_id":"2308.00562","n_code_links":0,"syntology":null},{"paper":null,"slug":"pixel-to-policy-dqn-encoders-for-within-cross","title":"Pixel to policy: DQN Encoders for within & cross-game reinforcement learning","date":"2023-08-01","arxiv_id":"2308.00318","n_code_links":0,"syntology":null},{"paper":null,"slug":"ether-aligning-emergent-communication-for","title":"ETHER: Aligning Emergent Communication for Hindsight Experience Replay","date":"2023-07-28","arxiv_id":"2307.15494","n_code_links":0,"syntology":null},{"paper":null,"slug":"goal-conditioned-reinforcement-learning-with-1","title":"Goal-Conditioned Reinforcement Learning with Disentanglement-based Reachability Planning","date":"2023-07-20","arxiv_id":"2307.10846","n_code_links":0,"syntology":null},{"paper":null,"slug":"distributed-3d-beam-reforming-for-hovering","title":"Distributed 3D-Beam Reforming for Hovering-Tolerant UAVs Communication over Coexistence: A Deep-Q Learning for Intelligent Space-Air-Ground Integrated Networks","date":"2023-07-18","arxiv_id":"2307.09325","n_code_links":0,"syntology":null},{"paper":null,"slug":"measuring-and-mitigating-interference-in-1","title":"Measuring and Mitigating Interference in Reinforcement Learning","date":"2023-07-10","arxiv_id":"2307.04887","n_code_links":0,"syntology":null},{"paper":"/paper/probabilistic-counterexample-guidance-for","slug":"probabilistic-counterexample-guidance-for","title":"Probabilistic Counterexample Guidance for Safer Reinforcement Learning (Extended Version)","date":"2023-07-10","arxiv_id":"2307.04927","n_code_links":1,"syntology":null},{"paper":null,"slug":"investigating-the-edge-of-stability","title":"Investigating the Edge of Stability Phenomenon in Reinforcement Learning","date":"2023-07-09","arxiv_id":"2307.04210","n_code_links":0,"syntology":null},{"paper":null,"slug":"incorporating-deep-q-network-with-multiclass","title":"Comparing Multiclass Classification Algorithms for Financial Distress Prediction","date":"2023-07-08","arxiv_id":"2307.03908","n_code_links":0,"syntology":null},{"paper":"/paper/containergym-a-real-world-reinforcement","slug":"containergym-a-real-world-reinforcement","title":"ContainerGym: A Real-World Reinforcement Learning Benchmark for Resource Allocation","date":"2023-07-06","arxiv_id":"2307.02991","n_code_links":1,"syntology":null},{"paper":null,"slug":"interpretable-and-secure-trajectory","title":"Interpretable and Secure Trajectory Optimization for UAV-Assisted Communication","date":"2023-07-05","arxiv_id":"2307.02002","n_code_links":0,"syntology":null},{"paper":null,"slug":"vanishing-bias-heuristic-guided-reinforcement","title":"Vanishing Bias Heuristic-guided Reinforcement Learning Algorithm","date":"2023-06-17","arxiv_id":"2306.10216","n_code_links":0,"syntology":null},{"paper":null,"slug":"quasi-newton-updating-for-large-scale","title":"Quasi-Newton Updating for Large-Scale Distributed Learning","date":"2023-06-07","arxiv_id":"2306.04111","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-q-learning-versus-proximal-policy","title":"Deep Q-Learning versus Proximal Policy Optimization: Performance Comparison in a Material Sorting Task","date":"2023-06-02","arxiv_id":"2306.01451","n_code_links":0,"syntology":null},{"paper":null,"slug":"va-learning-as-a-more-efficient-alternative","title":"VA-learning as a more efficient alternative to Q-learning","date":"2023-05-29","arxiv_id":"2305.18161","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-based-multi-1","title":"Deep Reinforcement Learning-based Multi-objective Path Planning on the Off-road Terrain Environment for Ground Vehicles","date":"2023-05-23","arxiv_id":"2305.13783","n_code_links":0,"syntology":null},{"paper":null,"slug":"how-does-agency-impact-human-ai-collaborative","title":"How does agency impact human-AI collaborative design space exploration? A case study on ship design with deep generative models","date":"2023-05-16","arxiv_id":"2305.10451","n_code_links":0,"syntology":null},{"paper":"/paper/an-intelligent-sdwn-routing-algorithm-based","slug":"an-intelligent-sdwn-routing-algorithm-based","title":"An Intelligent SDWN Routing Algorithm Based on Network Situational Awareness and Deep Reinforcement Learning","date":"2023-05-12","arxiv_id":"2305.10441","n_code_links":1,"syntology":null},{"paper":null,"slug":"on-practical-robust-reinforcement-learning","title":"On Practical Robust Reinforcement Learning: Practical Uncertainty Set and Double-Agent Algorithm","date":"2023-05-11","arxiv_id":"2305.06657","n_code_links":0,"syntology":null},{"paper":"/paper/extracting-diagnosis-pathways-from-electronic","slug":"extracting-diagnosis-pathways-from-electronic","title":"Extracting Diagnosis Pathways from Electronic Health Records Using Deep Reinforcement Learning","date":"2023-05-10","arxiv_id":"2305.06295","n_code_links":1,"syntology":null},{"paper":null,"slug":"improving-position-bias-estimation-against","title":"Position Bias Estimation with Item Embedding for Sparse Dataset","date":"2023-05-10","arxiv_id":"2305.13931","n_code_links":0,"syntology":null},{"paper":"/paper/bridging-rl-theory-and-practice-with-the-1","slug":"bridging-rl-theory-and-practice-with-the-1","title":"Bridging RL Theory and Practice with the Effective Horizon","date":"2023-04-19","arxiv_id":"2304.09853","n_code_links":1,"syntology":null},{"paper":"/paper/collaborative-multi-bs-power-management-for","slug":"collaborative-multi-bs-power-management-for","title":"Collaborative Multi-BS Power Management for Dense Radio Access Network using Deep Reinforcement Learning","date":"2023-04-17","arxiv_id":"2304.07976","n_code_links":1,"syntology":null},{"paper":null,"slug":"rels-dqn-a-robust-and-efficient-local-search","title":"RELS-DQN: A Robust and Efficient Local Search Framework for Combinatorial Optimization","date":"2023-04-11","arxiv_id":"2304.06048","n_code_links":0,"syntology":null},{"paper":null,"slug":"full-gradient-deep-reinforcement-learning-for","title":"Full Gradient Deep Reinforcement Learning for Average-Reward Criterion","date":"2023-04-07","arxiv_id":"2304.03729","n_code_links":0,"syntology":null},{"paper":null,"slug":"computational-role-of-sleep-in-memory","title":"Computational role of sleep in memory reorganization","date":"2023-04-06","arxiv_id":"2304.02873","n_code_links":0,"syntology":null},{"paper":"/paper/multi-agent-reinforcement-learning-with-6","slug":"multi-agent-reinforcement-learning-with-6","title":"Multi-Agent Reinforcement Learning with Action Masking for UAV-enabled Mobile Communications","date":"2023-03-29","arxiv_id":"2303.16737","n_code_links":1,"syntology":null},{"paper":null,"slug":"self-inspection-method-of-unmanned-aerial","title":"Self-Inspection Method of Unmanned Aerial Vehicles in Power Plants Using Deep Q-Network Reinforcement Learning","date":"2023-03-16","arxiv_id":"2303.09013","n_code_links":0,"syntology":null},{"paper":null,"slug":"recovering-arrhythmic-eeg-transients-from","title":"Recovering Arrhythmic EEG Transients from Their Stochastic Interference","date":"2023-03-14","arxiv_id":"2303.07683","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-framework-for-history-aware-hyperparameter","title":"A Framework for History-Aware Hyperparameter Optimisation in Reinforcement Learning","date":"2023-03-09","arxiv_id":"2303.05186","n_code_links":0,"syntology":null},{"paper":null,"slug":"provably-efficient-gauss-newton-temporal","title":"Gauss-Newton Temporal Difference Learning with Nonlinear Function Approximation","date":"2023-02-25","arxiv_id":"2302.13087","n_code_links":0,"syntology":null},{"paper":null,"slug":"ensemble-value-functions-for-efficient","title":"Ensemble Value Functions for Efficient Exploration in Multi-Agent Reinforcement Learning","date":"2023-02-07","arxiv_id":"2302.03439","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-for-traffic-light-1","title":"Deep Reinforcement Learning for Traffic Light Control in Intelligent Transportation Systems","date":"2023-02-04","arxiv_id":"2302.03669","n_code_links":0,"syntology":null},{"paper":null,"slug":"accelerating-policy-gradient-by-estimating","title":"Accelerating Policy Gradient by Estimating Value Function from Prior Computation in Deep Reinforcement Learning","date":"2023-02-02","arxiv_id":"2302.01399","n_code_links":0,"syntology":null},{"paper":null,"slug":"sample-efficient-deep-reinforcement-learning-3","title":"Sample Efficient Deep Reinforcement Learning via Local Planning","date":"2023-01-29","arxiv_id":"2301.12579","n_code_links":0,"syntology":null},{"paper":"/paper/schlably-a-python-framework-for-deep","slug":"schlably-a-python-framework-for-deep","title":"schlably: A Python Framework for Deep Reinforcement Learning Based Scheduling Experiments","date":"2023-01-10","arxiv_id":"2301.04182","n_code_links":1,"syntology":null},{"paper":null,"slug":"xdqn-inherently-interpretable-dqn-through","title":"XDQN: Inherently Interpretable DQN through Mimicking","date":"2023-01-08","arxiv_id":"2301.03043","n_code_links":0,"syntology":null},{"paper":null,"slug":"hierarchical-deep-reinforcement-learning-for","title":"Hierarchical Deep Reinforcement Learning for Age-of-Information Minimization in IRS-aided and Wireless-powered Wireless Networks","date":"2022-12-27","arxiv_id":"2212.13390","n_code_links":0,"syntology":null},{"paper":"/paper/control-of-continuous-quantum-systems-with","slug":"control-of-continuous-quantum-systems-with","title":"Control of Continuous Quantum Systems with Many Degrees of Freedom based on Convergent Reinforcement Learning","date":"2022-12-21","arxiv_id":"2212.10705","n_code_links":1,"syntology":null},{"paper":null,"slug":"neighboring-state-based-rl-exploration","title":"Neighboring state-based RL Exploration","date":"2022-12-21","arxiv_id":"2212.10712","n_code_links":0,"syntology":null},{"paper":null,"slug":"airfoil-shape-optimization-using-deep-q","title":"Airfoil Shape Optimization using Deep Q-Network","date":"2022-11-29","arxiv_id":"2211.17189","n_code_links":0,"syntology":null},{"paper":"/paper/applying-deep-reinforcement-learning-to-the","slug":"applying-deep-reinforcement-learning-to-the","title":"Applying Deep Reinforcement Learning to the HP Model for Protein Structure Prediction","date":"2022-11-27","arxiv_id":"2211.14939","n_code_links":1,"syntology":{"ran":11,"of":13,"n_ran_checked":11,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["compsoftmatterbiophysics-cityu-hk/applying-drl-to-hp-model-for-protein-structure-prediction"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"simultaneously-updating-all-persistence","title":"Simultaneously Updating All Persistence Values in Reinforcement Learning","date":"2022-11-21","arxiv_id":"2211.11620","n_code_links":0,"syntology":null},{"paper":null,"slug":"solar-power-driven-ev-charging-optimization","title":"Solar Power driven EV Charging Optimization with Deep Reinforcement Learning","date":"2022-11-17","arxiv_id":"2211.09479","n_code_links":0,"syntology":null},{"paper":null,"slug":"nrem-and-rem-cognitive-and-energetic-effects","title":"NREM and REM: cognitive and energetic gains in thalamo-cortical sleeping and awake spiking model","date":"2022-11-13","arxiv_id":"2211.06889","n_code_links":0,"syntology":null},{"paper":"/paper/deep-w-networks-solving-multi-objective","slug":"deep-w-networks-solving-multi-objective","title":"Deep W-Networks: Solving Multi-Objective Optimisation Problems With Deep Reinforcement Learning","date":"2022-11-09","arxiv_id":"2211.04813","n_code_links":1,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-for-power-control","title":"Deep Reinforcement Learning for Power Control in Next-Generation WiFi Network Systems","date":"2022-11-02","arxiv_id":"2211.01107","n_code_links":0,"syntology":null},{"paper":null,"slug":"representation-learning-for-general-sum-low","title":"Representation Learning for General-sum Low-rank Markov Games","date":"2022-10-30","arxiv_id":"2210.16976","n_code_links":0,"syntology":null},{"paper":"/paper/defix-detecting-and-fixing-failure-scenarios","slug":"defix-detecting-and-fixing-failure-scenarios","title":"DeFIX: Detecting and Fixing Failure Scenarios with Reinforcement Learning in Imitation Learning Based Autonomous Driving","date":"2022-10-29","arxiv_id":"2210.16567","n_code_links":2,"syntology":null},{"paper":null,"slug":"elastic-step-dqn-a-novel-multi-step-algorithm","title":"Elastic Step DQN: A novel multi-step algorithm to alleviate overestimation in Deep QNetworks","date":"2022-10-07","arxiv_id":"2210.03325","n_code_links":0,"syntology":null},{"paper":"/paper/scaling-directed-controller-synthesis-via","slug":"scaling-directed-controller-synthesis-via","title":"Exploration Policies for On-the-Fly Controller Synthesis: A Reinforcement Learning Approach","date":"2022-10-07","arxiv_id":"2210.05393","n_code_links":1,"syntology":null},{"paper":"/paper/m-2-dqn-a-robust-method-for-accelerating-deep","slug":"m-2-dqn-a-robust-method-for-accelerating-deep","title":"M$^2$DQN: A Robust Method for Accelerating Deep Q-learning Network","date":"2022-09-16","arxiv_id":"2209.07809","n_code_links":1,"syntology":null},{"paper":"/paper/reducing-variance-in-temporal-difference","slug":"reducing-variance-in-temporal-difference","title":"Reducing Variance in Temporal-Difference Value Estimation via Ensemble of Deep Networks","date":"2022-09-16","arxiv_id":"2209.07670","n_code_links":1,"syntology":null},{"paper":null,"slug":"pathfinding-in-random-partially-observable","title":"Pathfinding in Random Partially Observable Environments with Vision-Informed Deep Reinforcement Learning","date":"2022-09-11","arxiv_id":"2209.04801","n_code_links":0,"syntology":null},{"paper":null,"slug":"continual-learning-benefits-from-multiple","title":"Continual learning benefits from multiple sleep mechanisms: NREM, REM, and Synaptic Downscaling","date":"2022-09-09","arxiv_id":"2209.05245","n_code_links":0,"syntology":null},{"paper":null,"slug":"distilling-deep-rl-models-into-interpretable","title":"Distilling Deep RL Models Into Interpretable Neuro-Fuzzy Systems","date":"2022-09-07","arxiv_id":"2209.03357","n_code_links":0,"syntology":null},{"paper":"/paper/disentangled-modeling-of-domain-and-relevance","slug":"disentangled-modeling-of-domain-and-relevance","title":"Disentangled Modeling of Domain and Relevance for Adaptable Dense Retrieval","date":"2022-08-11","arxiv_id":"2208.05753","n_code_links":1,"syntology":null},{"paper":null,"slug":"prediction-based-hybrid-slicing-framework-for","title":"Prediction-based Hybrid Slicing Framework for Service Level Agreement Guarantee in Mobility Scenarios: A Deep Learning Approach","date":"2022-08-06","arxiv_id":"2208.03460","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-maintenance-planning-framework-using-online","title":"A Maintenance Planning Framework using Online and Offline Deep Reinforcement Learning","date":"2022-08-01","arxiv_id":"2208.00808","n_code_links":0,"syntology":null},{"paper":"/paper/a-deep-reinforcement-learning-approach-for-13","slug":"a-deep-reinforcement-learning-approach-for-13","title":"A Deep Reinforcement Learning Approach for Finding Non-Exploitable Strategies in Two-Player Atari Games","date":"2022-07-18","arxiv_id":"2207.08894","n_code_links":2,"syntology":{"ran":5,"of":10,"n_ran_checked":5,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["quantumiracle/mars","quantumiracle/nash-dqn"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"boolean-decision-rules-for-reinforcement","title":"Boolean Decision Rules for Reinforcement Learning Policy Summarisation","date":"2022-07-18","arxiv_id":"2207.08651","n_code_links":0,"syntology":null},{"paper":"/paper/deep-reinforcement-learning-with-swin","slug":"deep-reinforcement-learning-with-swin","title":"Deep Reinforcement Learning with Swin Transformers","date":"2022-06-30","arxiv_id":"2206.15269","n_code_links":1,"syntology":null},{"paper":"/paper/dna-proximal-policy-optimization-with-a-dual","slug":"dna-proximal-policy-optimization-with-a-dual","title":"DNA: Proximal Policy Optimization with a Dual Network Architecture","date":"2022-06-20","arxiv_id":"2206.10027","n_code_links":1,"syntology":null},{"paper":"/paper/sampling-efficient-deep-reinforcement","slug":"sampling-efficient-deep-reinforcement","title":"Sampling Efficient Deep Reinforcement Learning through Preference-Guided Stochastic Exploration","date":"2022-06-20","arxiv_id":"2206.09627","n_code_links":1,"syntology":null},{"paper":null,"slug":"goal-space-planning-with-subgoal-models","title":"Goal-Space Planning with Subgoal Models","date":"2022-06-06","arxiv_id":"2206.02902","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-phenomenon-of-policy-churn","title":"The Phenomenon of Policy Churn","date":"2022-06-01","arxiv_id":"2206.00730","n_code_links":0,"syntology":null},{"paper":"/paper/efficient-reward-poisoning-attacks-on-online","slug":"efficient-reward-poisoning-attacks-on-online","title":"Efficient Reward Poisoning Attacks on Online Deep Reinforcement Learning","date":"2022-05-30","arxiv_id":"2205.14842","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yinglunxu/reward_poisoning_attack_drl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"improving-bidding-and-playing-strategies-in","title":"Improving Bidding and Playing Strategies in the Trick-Taking game Wizard using Deep Q-Networks","date":"2022-05-27","arxiv_id":"2205.13834","n_code_links":0,"syntology":null},{"paper":null,"slug":"multiple-domain-cyberspace-attack-and-defense","title":"Multiple Domain Cyberspace Attack and Defense Game Based on Reward Randomization Reinforcement Learning","date":"2022-05-23","arxiv_id":"2205.10990","n_code_links":0,"syntology":null},{"paper":null,"slug":"long-run-incremental-cost-lric-distribution","title":"Long Run Incremental Cost (LRIC) Distribution Network Pricing in UK, advising China's Distribution Network","date":"2022-05-20","arxiv_id":"2205.09946","n_code_links":0,"syntology":null}],"record_sha256":"7a50bafce714baf3b864d914afffd1ab86863fd04da8349bf67b3a6d49a1f7c7","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}