{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/q-learning/papers/5","list_of":"/method/q-learning","method":"Q-Learning","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":5,"pages_in_order":18,"rows_per_page":100,"rows":[401,500],"of":1734,"counts":{"archive_papers_tagged":1734,"with_a_code_link":464,"where_syntology_ran_a_sample":126,"not_listed_spam_title":0,"listed":1734,"listed_where_code_ran":126,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":105,"every_run_a_failure_of_syntologys_instrument":21,"listed_with_a_run_with_no_instrument_failure":105,"listed_every_run_a_failure_of_syntologys_instrument":21,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/q-learning","prev":"/method/q-learning/papers/4","next":"/method/q-learning/papers/6","papers":[{"paper":null,"slug":"q-learning-based-optimal-false-data-injection","title":"Q-learning Based Optimal False Data Injection Attack on Probabilistic Boolean Control Networks","date":"2023-11-29","arxiv_id":"2311.17631","n_code_links":0,"syntology":null},{"paper":null,"slug":"self-driving-telescopes-autonomous-scheduling","title":"Self-Driving Telescopes: Autonomous Scheduling of Astronomical Observation Campaigns with Offline Reinforcement Learning","date":"2023-11-29","arxiv_id":"2311.18094","n_code_links":0,"syntology":null},{"paper":"/paper/reinforcement-learning-for-wildfire","slug":"reinforcement-learning-for-wildfire","title":"Reinforcement Learning for Wildfire Mitigation in Simulated Disaster Environments","date":"2023-11-27","arxiv_id":"2311.15925","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":5,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["mitrefireline/simfire"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"reinforcement-learning-from-diffusion","title":"Reinforcement Learning from Diffusion Feedback: Q* for Image Search","date":"2023-11-27","arxiv_id":"2311.15648","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-nearly-optimal-and-low-switching-algorithm","title":"A Nearly Optimal and Low-Switching Algorithm for Reinforcement Learning with General Function Approximation","date":"2023-11-26","arxiv_id":"2311.15238","n_code_links":0,"syntology":null},{"paper":null,"slug":"projected-off-policy-q-learning-pop-ql-for","title":"Projected Off-Policy Q-Learning (POP-QL) for Stabilizing Offline Reinforcement Learning","date":"2023-11-25","arxiv_id":"2311.14885","n_code_links":0,"syntology":null},{"paper":null,"slug":"approximation-of-convex-envelope-using","title":"Approximation of Convex Envelope Using Reinforcement Learning","date":"2023-11-24","arxiv_id":"2311.14421","n_code_links":0,"syntology":null},{"paper":null,"slug":"efficient-open-world-reinforcement-learning","title":"Efficient Open-world Reinforcement Learning via Knowledge Distillation and Autonomous Rule Discovery","date":"2023-11-24","arxiv_id":"2311.14270","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-optimal-tracking-portfolio-in-incomplete","title":"On optimal tracking portfolio in incomplete markets: The reinforcement learning approach","date":"2023-11-24","arxiv_id":"2311.14318","n_code_links":0,"syntology":null},{"paper":"/paper/l-m-v-iql-multiple-intention-inverse","slug":"l-m-v-iql-multiple-intention-inverse","title":"Multi-intention Inverse Q-learning for Interpretable Behavior Representation","date":"2023-11-23","arxiv_id":"2311.13870","n_code_links":1,"syntology":null},{"paper":null,"slug":"machine-learning-based-decentralized-tdma-for","title":"Machine learning-based decentralized TDMA for VLC IoT networks","date":"2023-11-23","arxiv_id":"2311.14078","n_code_links":0,"syntology":null},{"paper":null,"slug":"decentralised-q-learning-for-multi-agent","title":"Decentralised Q-Learning for Multi-Agent Markov Decision Processes with a Satisfiability Criterion","date":"2023-11-21","arxiv_id":"2311.12613","n_code_links":0,"syntology":null},{"paper":null,"slug":"offline-reinforcement-learning-for-wireless","title":"Offline Reinforcement Learning for Wireless Network Optimization with Mixture Datasets","date":"2023-11-19","arxiv_id":"2311.11423","n_code_links":0,"syntology":null},{"paper":null,"slug":"genetic-algorithm-enhanced-by-deep","title":"Genetic Algorithm enhanced by Deep Reinforcement Learning in parent selection mechanism and mutation : Minimizing makespan in permutation flow shop scheduling problems","date":"2023-11-10","arxiv_id":"2311.05937","n_code_links":0,"syntology":null},{"paper":null,"slug":"two-compartment-neuronal-spiking-model","title":"Two-compartment neuronal spiking model expressing brain-state specific apical-amplification, -isolation and -drive regimes","date":"2023-11-10","arxiv_id":"2311.06074","n_code_links":0,"syntology":null},{"paper":null,"slug":"advancing-algorithmic-trading-a-multi","title":"Advancing Algorithmic Trading: A Multi-Technique Enhancement of Deep Q-Network Models","date":"2023-11-09","arxiv_id":"2311.05743","n_code_links":0,"syntology":null},{"paper":"/paper/selectively-sharing-experiences-improves","slug":"selectively-sharing-experiences-improves","title":"Selectively Sharing Experiences Improves Multi-Agent Reinforcement Learning","date":"2023-11-01","arxiv_id":"2311.00865","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["mgerstgrasser/super"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"q-learning-for-stochastic-control-under","title":"Q-Learning for Stochastic Control under General Information Structures and Non-Markovian Environments","date":"2023-10-31","arxiv_id":"2311.00123","n_code_links":0,"syntology":null},{"paper":null,"slug":"dgfn-double-generative-flow-networks","title":"DGFN: Double Generative Flow Networks","date":"2023-10-30","arxiv_id":"2310.19685","n_code_links":0,"syntology":null},{"paper":null,"slug":"automaton-distillation-neuro-symbolic","title":"Automaton Distillation: Neuro-Symbolic Transfer Learning for Deep Reinforcement Learning","date":"2023-10-29","arxiv_id":"2310.19137","n_code_links":0,"syntology":null},{"paper":"/paper/weakly-coupled-deep-q-networks","slug":"weakly-coupled-deep-q-networks","title":"Weakly Coupled Deep Q-Networks","date":"2023-10-28","arxiv_id":"2310.18803","n_code_links":0,"syntology":{"ran":5,"of":10,"n_ran_checked":3,"n_instrument":2,"unverified":5,"pointer_only":10,"phrase":"5 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","official":null}},{"paper":null,"slug":"lifting-the-veil-unlocking-the-power-of-depth","title":"Lifting the Veil: Unlocking the Power of Depth in Q-learning","date":"2023-10-27","arxiv_id":"2310.17915","n_code_links":0,"syntology":null},{"paper":null,"slug":"model-free-posterior-sampling-via-learning","title":"Model-free Posterior Sampling via Learning Rate Randomization","date":"2023-10-27","arxiv_id":"2310.18186","n_code_links":0,"syntology":null},{"paper":null,"slug":"integrated-freeway-and-arterial-traffic","title":"Integrated Freeway Traffic Control Using Q-Learning with Adjacent Arterial Traffic Considerations","date":"2023-10-25","arxiv_id":"2310.16748","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-convergence-and-sample-complexity","title":"On the Convergence and Sample Complexity Analysis of Deep Q-Networks with $ε$-Greedy Exploration","date":"2023-10-24","arxiv_id":"2310.16173","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-based-local-path","title":"Reinforcement learning based local path planning for mobile robot","date":"2023-10-24","arxiv_id":"2403.12463","n_code_links":0,"syntology":null},{"paper":null,"slug":"ai-on-the-water-applying-drl-to-autonomous","title":"AI on the Water: Applying DRL to Autonomous Vessel Navigation","date":"2023-10-23","arxiv_id":"2310.14938","n_code_links":0,"syntology":null},{"paper":"/paper/deep-reinforcement-learning-based-intelligent-2","slug":"deep-reinforcement-learning-based-intelligent-2","title":"Deep Reinforcement Learning-based Intelligent Traffic Signal Controls with Optimized CO2 emissions","date":"2023-10-19","arxiv_id":"2310.13129","n_code_links":1,"syntology":null},{"paper":"/paper/towards-robust-offline-reinforcement-learning","slug":"towards-robust-offline-reinforcement-learning","title":"Towards Robust Offline Reinforcement Learning under Diverse Data Corruption","date":"2023-10-19","arxiv_id":"2310.12955","n_code_links":2,"syntology":{"ran":5,"of":9,"n_ran_checked":3,"n_instrument":2,"unverified":4,"pointer_only":9,"phrase":"5 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","official":{"repos":["yangrui2015/riql"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["listed","official","unlocated"]}}},{"paper":null,"slug":"automatic-music-playlist-generation-via","title":"Automatic Music Playlist Generation via Simulation-based Reinforcement Learning","date":"2023-10-13","arxiv_id":"2310.09123","n_code_links":0,"syntology":null},{"paper":null,"slug":"optimal-scheduling-of-electric-vehicle","title":"Optimal Scheduling of Electric Vehicle Charging with Deep Reinforcement Learning considering End Users Flexibility","date":"2023-10-13","arxiv_id":"2310.09040","n_code_links":0,"syntology":null},{"paper":null,"slug":"when-are-bandits-robust-to-misspecification","title":"Bad Values but Good Behavior: Learning Highly Misspecified Bandits and MDPs","date":"2023-10-13","arxiv_id":"2310.09358","n_code_links":0,"syntology":null},{"paper":"/paper/learning-rl-policies-for-joint-beamforming","slug":"learning-rl-policies-for-joint-beamforming","title":"Learning RL-Policies for Joint Beamforming Without Exploration: A Batch Constrained Off-Policy Approach","date":"2023-10-12","arxiv_id":"2310.08660","n_code_links":1,"syntology":null},{"paper":null,"slug":"integrated-sensing-and-communication-neighbor","title":"Integrated Sensing and Communication Neighbor Discovery for MANET with Gossip Mechanism","date":"2023-10-11","arxiv_id":"2310.07292","n_code_links":0,"syntology":null},{"paper":"/paper/boosting-continuous-control-with-consistency","slug":"boosting-continuous-control-with-consistency","title":"Boosting Continuous Control with Consistency Policy","date":"2023-10-10","arxiv_id":"2310.06343","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["cccedric/cpql"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"suppressing-overestimation-in-q-learning","title":"Suppressing Overestimation in Q-Learning through Adversarial Behaviors","date":"2023-10-10","arxiv_id":"2310.06286","n_code_links":0,"syntology":null},{"paper":"/paper/deepqtest-testing-autonomous-driving-systems","slug":"deepqtest-testing-autonomous-driving-systems","title":"DeepQTest: Testing Autonomous Driving Systems with Reinforcement Learning and Real-world Weather Data","date":"2023-10-08","arxiv_id":"2310.05170","n_code_links":1,"syntology":null},{"paper":null,"slug":"digital-twin-assisted-deep-reinforcement","title":"Digital Twin Assisted Deep Reinforcement Learning for Online Admission Control in Sliced Network","date":"2023-10-07","arxiv_id":"2310.09299","n_code_links":0,"syntology":null},{"paper":null,"slug":"optimal-sequential-decision-making-in","title":"Optimal Sequential Decision-Making in Geosteering: A Reinforcement Learning Approach","date":"2023-10-07","arxiv_id":"2310.04772","n_code_links":0,"syntology":null},{"paper":null,"slug":"applying-reinforcement-learning-to-option","title":"Applying Reinforcement Learning to Option Pricing and Hedging","date":"2023-10-06","arxiv_id":"2310.04336","n_code_links":0,"syntology":null},{"paper":null,"slug":"optimal-control-of-district-cooling-energy","title":"Optimal Control of District Cooling Energy Plant with Reinforcement Learning and MPC","date":"2023-10-05","arxiv_id":"2310.03814","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-deep-reinforcement-learning-approach-for-15","title":"A Deep Reinforcement Learning Approach for Interactive Search with Sentence-level Feedback","date":"2023-10-03","arxiv_id":"2310.03043","n_code_links":0,"syntology":null},{"paper":null,"slug":"finite-time-analysis-of-whittle-index-based-q","title":"Finite-Time Analysis of Whittle Index based Q-Learning for Restless Multi-Armed Bandits with Neural Network Function Approximation","date":"2023-10-03","arxiv_id":"2310.02147","n_code_links":0,"syntology":null},{"paper":"/paper/pgdqn-preference-guided-deep-q-network","slug":"pgdqn-preference-guided-deep-q-network","title":"PGDQN: Preference-Guided Deep Q-Network","date":"2023-10-03","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"using-reinforcement-learning-to-optimize","title":"Using Reinforcement Learning to Optimize Responses in Care Processes: A Case Study on Aggression Incidents","date":"2023-10-02","arxiv_id":"2310.00981","n_code_links":0,"syntology":null},{"paper":"/paper/pre-training-with-synthetic-data-helps","slug":"pre-training-with-synthetic-data-helps","title":"Pre-training with Synthetic Data Helps Offline Reinforcement Learning","date":"2023-10-01","arxiv_id":"2310.00771","n_code_links":1,"syntology":{"ran":7,"of":7,"n_ran_checked":7,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["victor-wang-902/synthetic-pretrain-rl"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"reinforcement-learning-adaptive-fuzzy","title":"Reinforcement learning adaptive fuzzy controller for lighting systems: application to aircraft cabin","date":"2023-09-30","arxiv_id":"2310.00525","n_code_links":0,"syntology":null},{"paper":null,"slug":"universal-sleep-decoder-aligning-awake-and","title":"SI-SD: Sleep Interpreter through awake-guided cross-subject Semantic Decoding","date":"2023-09-28","arxiv_id":"2309.16457","n_code_links":0,"syntology":null},{"paper":null,"slug":"decoding-trust-a-reinforcement-learning","title":"Decoding trust: A reinforcement learning perspective","date":"2023-09-26","arxiv_id":"2309.14598","n_code_links":0,"syntology":null},{"paper":null,"slug":"adapting-double-q-learning-for-continuous","title":"Adapting Double Q-Learning for Continuous Reinforcement Learning","date":"2023-09-25","arxiv_id":"2309.14471","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-for-the-heat","title":"Deep Reinforcement Learning for the Heat Transfer Control of Pulsating Impinging Jets","date":"2023-09-25","arxiv_id":"2309.13955","n_code_links":0,"syntology":null},{"paper":"/paper/enhancing-data-efficiency-in-reinforcement","slug":"enhancing-data-efficiency-in-reinforcement","title":"Enhancing data efficiency in reinforcement learning: a novel imagination mechanism based on mesh information propagation","date":"2023-09-25","arxiv_id":"2309.14243","n_code_links":2,"syntology":null},{"paper":null,"slug":"enhancing-healthcare-with-eog-a-novel","title":"Enhancing Healthcare with EOG: A Novel Approach to Sleep Stage Classification","date":"2023-09-25","arxiv_id":"2310.03757","n_code_links":0,"syntology":null},{"paper":null,"slug":"implicit-sensing-in-traffic-optimization","title":"Implicit Sensing in Traffic Optimization: Advanced Deep Reinforcement Learning Techniques","date":"2023-09-25","arxiv_id":"2309.14395","n_code_links":0,"syntology":null},{"paper":"/paper/counterfactual-conservative-q-learning-for-1","slug":"counterfactual-conservative-q-learning-for-1","title":"Counterfactual Conservative Q Learning for Offline Multi-agent Reinforcement Learning","date":"2023-09-22","arxiv_id":"2309.12696","n_code_links":1,"syntology":{"ran":7,"of":9,"n_ran_checked":4,"n_instrument":3,"unverified":2,"pointer_only":9,"phrase":"7 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["thu-rllab/CFCQL"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/belief-projection-based-reinforcement","slug":"belief-projection-based-reinforcement","title":"Belief Projection-Based Reinforcement Learning for Environments with Delayed Feedback","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/double-gumbel-q-learning","slug":"double-gumbel-q-learning","title":"Double Gumbel Q-Learning","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"on-the-convergence-and-sample-complexity-1","title":"On the Convergence and Sample Complexity Analysis of Deep Q-Networks with $\\epsilon$-Greedy Exploration","date":"2023-09-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"reds-offline-rl-with-heteroskedastic-datasets","title":"ReDS: Offline RL With Heteroskedastic Datasets via Support Constraints","date":"2023-09-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"uav-swarm-deployment-and-trajectory-for-3d","title":"UAV Swarm Deployment and Trajectory for 3D Area Coverage via Reinforcement Learning","date":"2023-09-21","arxiv_id":"2309.11992","n_code_links":0,"syntology":null},{"paper":null,"slug":"ai-driven-patient-monitoring-with-multi-agent","title":"Adaptive Multi-Agent Deep Reinforcement Learning for Timely Healthcare Interventions","date":"2023-09-20","arxiv_id":"2309.10980","n_code_links":0,"syntology":null},{"paper":null,"slug":"differentiable-quantum-architecture-search-1","title":"Differentiable Quantum Architecture Search for Quantum Reinforcement Learning","date":"2023-09-19","arxiv_id":"2309.10392","n_code_links":0,"syntology":null},{"paper":null,"slug":"double-deep-q-learning-based-path-selection","title":"Double Deep Q-Learning-based Path Selection and Service Placement for Latency-Sensitive Beyond 5G Applications","date":"2023-09-18","arxiv_id":"2309.10180","n_code_links":0,"syntology":null},{"paper":null,"slug":"self-sustaining-multiple-access-with-1","title":"Self-Sustaining Multiple Access with Continual Deep Reinforcement Learning for Dynamic Metaverse Applications","date":"2023-09-18","arxiv_id":"2309.10177","n_code_links":0,"syntology":null},{"paper":"/paper/dynamic-control-of-self-assembly-of","slug":"dynamic-control-of-self-assembly-of","title":"Dynamic control of self-assembly of quasicrystalline structures through reinforcement learning","date":"2023-09-13","arxiv_id":"2309.06869","n_code_links":2,"syntology":null},{"paper":null,"slug":"a-q-learning-approach-for-adherence-aware","title":"A Q-learning Approach for Adherence-Aware Recommendations","date":"2023-09-12","arxiv_id":"2309.06519","n_code_links":0,"syntology":null},{"paper":null,"slug":"career-path-recommendations-for-long-term","title":"Career Path Recommendations for Long-term Income Maximization: A Reinforcement Learning Approach","date":"2023-09-11","arxiv_id":"2309.05391","n_code_links":0,"syntology":null},{"paper":null,"slug":"convex-q-learning-in-a-stochastic-environment","title":"Convex Q Learning in a Stochastic Environment: Extended Version","date":"2023-09-10","arxiv_id":"2309.05105","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-agent-deeprl-based-joint-power-and","title":"Multi Agent DeepRL based Joint Power and Subchannel Allocation in IAB networks","date":"2023-08-31","arxiv_id":"2309.00144","n_code_links":0,"syntology":null},{"paper":null,"slug":"coalescent-processes-emerging-from-large","title":"Coalescent processes emerging from large deviations","date":"2023-08-28","arxiv_id":"2308.14715","n_code_links":0,"syntology":null},{"paper":"/paper/learning-visual-tracking-and-reaching-with","slug":"learning-visual-tracking-and-reaching-with","title":"Learning Visual Tracking and Reaching with Deep Reinforcement Learning on a UR10e Robotic Arm","date":"2023-08-28","arxiv_id":"2308.14652","n_code_links":1,"syntology":null},{"paper":"/paper/llm-powered-sim-to-real-transfer-for-traffic","slug":"llm-powered-sim-to-real-transfer-for-traffic","title":"Prompt to Transfer: Sim-to-Real Transfer for Traffic Signal Control with Prompt Learning","date":"2023-08-28","arxiv_id":"2308.14284","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["darl-libsignal/promptgat"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/reinforcement-learning-for-sampling-on","slug":"reinforcement-learning-for-sampling-on","title":"Reinforcement Learning for Sampling on Temporal Medical Imaging Sequences","date":"2023-08-28","arxiv_id":"2308.14946","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["zhishenhuang/rlsamp"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"understanding-the-usage-of-qubo-based","title":"A Graph Neural Network-Based QUBO-Formulated Hamiltonian-Inspired Loss Function for Combinatorial Optimization using Reinforcement Learning","date":"2023-08-27","arxiv_id":"2308.13978","n_code_links":0,"syntology":null},{"paper":null,"slug":"actuator-trajectory-planning-for-uavs-with","title":"Actuator Trajectory Planning for UAVs with Overhead Manipulator using Reinforcement Learning","date":"2023-08-24","arxiv_id":"2308.12843","n_code_links":0,"syntology":null},{"paper":"/paper/towards-few-shot-coordination-revisiting-ad","slug":"towards-few-shot-coordination-revisiting-ad","title":"Towards Few-shot Coordination: Revisiting Ad-hoc Teamplay Challenge In the Game of Hanabi","date":"2023-08-20","arxiv_id":"2308.10284","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["chandar-lab/adaptive-hanabi"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"model-free-algorithm-with-improved-sample","title":"Improving Sample Efficiency of Model-Free Algorithms for Zero-Sum Markov Games","date":"2023-08-17","arxiv_id":"2308.08858","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-for-battery-management","title":"Reinforcement Learning for Battery Management in Dairy Farming","date":"2023-08-17","arxiv_id":"2308.09023","n_code_links":0,"syntology":null},{"paper":null,"slug":"integrating-renewable-energy-in-agriculture-a","title":"Integrating Renewable Energy in Agriculture: A Deep Reinforcement Learning-based Approach","date":"2023-08-16","arxiv_id":"2308.08611","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-rl-augmented-cold","title":"On-demand Cold Start Frequency Reduction with Off-Policy Reinforcement Learning in Serverless Computing","date":"2023-08-15","arxiv_id":"2308.07541","n_code_links":0,"syntology":null},{"paper":"/paper/variations-on-the-reinforcement-learning","slug":"variations-on-the-reinforcement-learning","title":"Variations on the Reinforcement Learning performance of Blackjack","date":"2023-08-09","arxiv_id":"2308.07329","n_code_links":2,"syntology":null},{"paper":null,"slug":"asynchronous-decentralized-q-learning-two","title":"Unsynchronized Decentralized Q-Learning: Two Timescale Analysis By Persistence","date":"2023-08-07","arxiv_id":"2308.03239","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-q-network-for-stochastic-process","title":"Deep Q-Network for Stochastic Process Environments","date":"2023-08-07","arxiv_id":"2308.03316","n_code_links":0,"syntology":null},{"paper":null,"slug":"bag-of-policies-for-distributional-deep","title":"Bag of Policies for Distributional Deep Exploration","date":"2023-08-03","arxiv_id":"2308.01759","n_code_links":0,"syntology":null},{"paper":null,"slug":"caching-at-stars-the-next-generation-edge","title":"Caching-at-STARS: the Next Generation Edge Caching","date":"2023-08-01","arxiv_id":"2308.00562","n_code_links":0,"syntology":null},{"paper":null,"slug":"pixel-to-policy-dqn-encoders-for-within-cross","title":"Pixel to policy: DQN Encoders for within & cross-game reinforcement learning","date":"2023-08-01","arxiv_id":"2308.00318","n_code_links":0,"syntology":null},{"paper":"/paper/robust-multi-agent-reinforcement-learning-3","slug":"robust-multi-agent-reinforcement-learning-3","title":"Robust Multi-Agent Reinforcement Learning with State Uncertainty","date":"2023-07-30","arxiv_id":"2307.16212","n_code_links":1,"syntology":{"ran":2,"of":5,"n_ran_checked":2,"n_instrument":0,"unverified":3,"pointer_only":5,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["sihongho/robust_marl_with_state_uncertainty"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"ether-aligning-emergent-communication-for","title":"ETHER: Aligning Emergent Communication for Hindsight Experience Replay","date":"2023-07-28","arxiv_id":"2307.15494","n_code_links":0,"syntology":null},{"paper":null,"slug":"stability-of-multi-agent-learning-convergence","title":"Stability of Multi-Agent Learning: Convergence in Network Games with Many Players","date":"2023-07-26","arxiv_id":"2307.13922","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-flexible-framework-for-incorporating","title":"A Flexible Framework for Incorporating Patient Preferences Into Q-Learning","date":"2023-07-22","arxiv_id":"2307.12022","n_code_links":0,"syntology":null},{"paper":"/paper/exploring-reinforcement-learning-techniques","slug":"exploring-reinforcement-learning-techniques","title":"Exploring reinforcement learning techniques for discrete and continuous control tasks in the MuJoCo environment","date":"2023-07-20","arxiv_id":"2307.11166","n_code_links":1,"syntology":null},{"paper":null,"slug":"goal-conditioned-reinforcement-learning-with-1","title":"Goal-Conditioned Reinforcement Learning with Disentanglement-based Reachability Planning","date":"2023-07-20","arxiv_id":"2307.10846","n_code_links":0,"syntology":null},{"paper":null,"slug":"distributed-3d-beam-reforming-for-hovering","title":"Distributed 3D-Beam Reforming for Hovering-Tolerant UAVs Communication over Coexistence: A Deep-Q Learning for Intelligent Space-Air-Ground Integrated Networks","date":"2023-07-18","arxiv_id":"2307.09325","n_code_links":0,"syntology":null},{"paper":"/paper/meta-value-learning-a-general-framework-for","slug":"meta-value-learning-a-general-framework-for","title":"Meta-Value Learning: a General Framework for Learning with Learning Awareness","date":"2023-07-17","arxiv_id":"2307.08863","n_code_links":1,"syntology":null},{"paper":null,"slug":"credit-assignment-challenges-and","title":"Credit Assignment: Challenges and Opportunities in Developing Human-like AI Agents","date":"2023-07-16","arxiv_id":"2307.08171","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-for-the-dynamic","title":"Deep reinforcement learning for the dynamic vehicle dispatching problem: An event-based approach","date":"2023-07-13","arxiv_id":"2307.07508","n_code_links":0,"syntology":null},{"paper":null,"slug":"realtime-spectrum-monitoring-via","title":"Realtime Spectrum Monitoring via Reinforcement Learning -- A Comparison Between Q-Learning and Heuristic Methods","date":"2023-07-11","arxiv_id":"2307.05763","n_code_links":0,"syntology":null},{"paper":null,"slug":"measuring-and-mitigating-interference-in-1","title":"Measuring and Mitigating Interference in Reinforcement Learning","date":"2023-07-10","arxiv_id":"2307.04887","n_code_links":0,"syntology":null},{"paper":"/paper/probabilistic-counterexample-guidance-for","slug":"probabilistic-counterexample-guidance-for","title":"Probabilistic Counterexample Guidance for Safer Reinforcement Learning (Extended Version)","date":"2023-07-10","arxiv_id":"2307.04927","n_code_links":1,"syntology":null},{"paper":null,"slug":"investigating-the-edge-of-stability","title":"Investigating the Edge of Stability Phenomenon in Reinforcement Learning","date":"2023-07-09","arxiv_id":"2307.04210","n_code_links":0,"syntology":null}],"record_sha256":"86a4bbb4ff0a924dfa5eae7737077880194b69ab0b2a5b7d6733b69abee26dfe","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}