{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/q-learning/papers/11","list_of":"/method/q-learning","method":"Q-Learning","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":11,"pages_in_order":18,"rows_per_page":100,"rows":[1001,1100],"of":1734,"counts":{"archive_papers_tagged":1734,"with_a_code_link":464,"where_syntology_ran_a_sample":126,"not_listed_spam_title":0,"listed":1734,"listed_where_code_ran":126,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":105,"every_run_a_failure_of_syntologys_instrument":21,"listed_with_a_run_with_no_instrument_failure":105,"listed_every_run_a_failure_of_syntologys_instrument":21,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/q-learning","prev":"/method/q-learning/papers/10","next":"/method/q-learning/papers/12","papers":[{"paper":"/paper/deep-reinforcement-learning-for-optimal-3","slug":"deep-reinforcement-learning-for-optimal-3","title":"Deep Reinforcement Learning for Optimal Stopping with Application in Financial Engineering","date":"2021-05-19","arxiv_id":"2105.08877","n_code_links":1,"syntology":null},{"paper":"/paper/improved-exploring-starts-by-kernel-density","slug":"improved-exploring-starts-by-kernel-density","title":"Improved Exploring Starts by Kernel Density Estimation-Based State-Space Coverage Acceleration in Reinforcement Learning","date":"2021-05-19","arxiv_id":"2105.08990","n_code_links":1,"syntology":null},{"paper":null,"slug":"efficient-off-policy-q-learning-for-data","title":"Efficient Off-Policy Q-Learning for Data-Based Discrete-Time LQR Problems","date":"2021-05-17","arxiv_id":"2105.07761","n_code_links":0,"syntology":null},{"paper":null,"slug":"learn-to-intervene-an-adaptive-learning","title":"Learn to Intervene: An Adaptive Learning Policy for Restless Bandits in Application to Preventive Healthcare","date":"2021-05-17","arxiv_id":"2105.07965","n_code_links":0,"syntology":null},{"paper":"/paper/uncertainty-weighted-actor-critic-for-offline","slug":"uncertainty-weighted-actor-critic-for-offline","title":"Uncertainty Weighted Actor-Critic for Offline Reinforcement Learning","date":"2021-05-17","arxiv_id":"2105.08140","n_code_links":2,"syntology":null},{"paper":null,"slug":"interpretable-performance-analysis-towards","title":"Interpretable performance analysis towards offline reinforcement learning: A dataset perspective","date":"2021-05-12","arxiv_id":"2105.05473","n_code_links":0,"syntology":null},{"paper":null,"slug":"fast-constraint-satisfaction-problem-and","title":"Fast constraint satisfaction problem and learning-based algorithm for solving Minesweeper","date":"2021-05-10","arxiv_id":"2105.04120","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-with-expert-trajectory","title":"Reinforcement Learning with Expert Trajectory For Quantitative Trading","date":"2021-05-09","arxiv_id":"2105.03844","n_code_links":0,"syntology":null},{"paper":null,"slug":"time-aware-q-networks-resolving-temporal","title":"Time-Aware Q-Networks: Resolving Temporal Irregularity for Deep Reinforcement Learning","date":"2021-05-06","arxiv_id":"2105.02580","n_code_links":0,"syntology":null},{"paper":"/paper/automated-scoring-of-pre-rem-sleep-in-mice","slug":"automated-scoring-of-pre-rem-sleep-in-mice","title":"Automated scoring of pre-REM sleep in mice with deep learning","date":"2021-05-05","arxiv_id":"2105.01933","n_code_links":1,"syntology":null},{"paper":null,"slug":"survey-on-multi-agent-q-learning-frameworks","title":"Survey on Multi-Agent Q-Learning frameworks for resource management in wireless sensor network","date":"2021-05-05","arxiv_id":"2105.02371","n_code_links":0,"syntology":null},{"paper":"/paper/hasco-towards-agile-hardware-and-software-co","slug":"hasco-towards-agile-hardware-and-software-co","title":"HASCO: Towards Agile HArdware and Software CO-design for Tensor Computation","date":"2021-05-04","arxiv_id":"2105.01585","n_code_links":1,"syntology":null},{"paper":"/paper/action-candidate-based-clipped-double-q","slug":"action-candidate-based-clipped-double-q","title":"Action Candidate Based Clipped Double Q-learning for Discrete and Continuous Action Tasks","date":"2021-05-03","arxiv_id":"2105.00704","n_code_links":1,"syntology":null},{"paper":"/paper/robotic-surgery-with-lean-reinforcement","slug":"robotic-surgery-with-lean-reinforcement","title":"Robotic Surgery With Lean Reinforcement Learning","date":"2021-05-03","arxiv_id":"2105.01006","n_code_links":1,"syntology":null},{"paper":null,"slug":"carl-dtn-context-adaptive-reinforcement","title":"CARL-DTN: Context Adaptive Reinforcement Learning based Routing Algorithm in Delay Tolerant Network","date":"2021-05-02","arxiv_id":"2105.00544","n_code_links":0,"syntology":null},{"paper":"/paper/adapting-to-reward-progressivity-via-spectral-1","slug":"adapting-to-reward-progressivity-via-spectral-1","title":"Adapting to Reward Progressivity via Spectral Reinforcement Learning","date":"2021-04-29","arxiv_id":"2104.14138","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","official":{"repos":["mchldann/SpectralDQN"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"antagonistic-crowd-simulation-model","title":"Emotional Contagion-Aware Deep Reinforcement Learning for Antagonistic Crowd Simulation","date":"2021-04-29","arxiv_id":"2105.00854","n_code_links":0,"syntology":null},{"paper":null,"slug":"rp-dqn-an-application-of-q-learning-to","title":"RP-DQN: An application of Q-Learning to Vehicle Routing Problems","date":"2021-04-25","arxiv_id":"2104.12226","n_code_links":0,"syntology":null},{"paper":"/paper/independent-reinforcement-learning-for-weakly","slug":"independent-reinforcement-learning-for-weakly","title":"Independent Reinforcement Learning for Weakly Cooperative Multiagent Traffic Control Problem","date":"2021-04-22","arxiv_id":"2104.10917","n_code_links":1,"syntology":null},{"paper":null,"slug":"model-aided-deep-reinforcement-learning-for","title":"Model-aided Deep Reinforcement Learning for Sample-efficient UAV Trajectory Design in IoT Networks","date":"2021-04-21","arxiv_id":"2104.10403","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-for-traffic-signal","title":"Reinforcement Learning for Traffic Signal Control: Comparison with Commercial Systems","date":"2021-04-21","arxiv_id":"2104.10455","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-simulated-experiment-to-explore-robotic","title":"A Simulated Experiment to Explore Robotic Dialogue Strategies for People with Dementia","date":"2021-04-18","arxiv_id":"2104.08940","n_code_links":0,"syntology":null},{"paper":"/paper/reinforcement-learning-based-process","slug":"reinforcement-learning-based-process","title":"Reinforcement learning based process optimization and strategy development in conventional tunneling","date":"2021-04-17","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"actionable-models-unsupervised-offline","title":"Actionable Models: Unsupervised Offline Reinforcement Learning of Robotic Skills","date":"2021-04-15","arxiv_id":"2104.07749","n_code_links":0,"syntology":null},{"paper":"/paper/a-coevolutionairy-approach-to-deep-multi","slug":"a-coevolutionairy-approach-to-deep-multi","title":"A coevolutionary approach to deep multi-agent reinforcement learning","date":"2021-04-12","arxiv_id":"2104.05610","n_code_links":1,"syntology":null},{"paper":null,"slug":"prospect-theoretic-q-learning","title":"Prospect-theoretic Q-learning","date":"2021-04-12","arxiv_id":"2104.05311","n_code_links":0,"syntology":null},{"paper":"/paper/group-equivariant-neural-architecture-search","slug":"group-equivariant-neural-architecture-search","title":"Autoequivariant Network Search via Group Decomposition","date":"2021-04-10","arxiv_id":"2104.04848","n_code_links":1,"syntology":null},{"paper":"/paper/optimal-market-making-by-reinforcement","slug":"optimal-market-making-by-reinforcement","title":"Optimal Market Making by Reinforcement Learning","date":"2021-04-08","arxiv_id":"2104.04036","n_code_links":1,"syntology":null},{"paper":null,"slug":"distributed-deep-reinforcement-learning-for-3","title":"Distributed Deep Reinforcement Learning for Collaborative Spectrum Sharing","date":"2021-04-06","arxiv_id":"2104.02059","n_code_links":0,"syntology":null},{"paper":null,"slug":"solo-search-online-learn-offline-for","title":"SOLO: Search Online, Learn Offline for Combinatorial Optimization Problems","date":"2021-04-04","arxiv_id":"2104.01646","n_code_links":0,"syntology":null},{"paper":null,"slug":"convergence-of-finite-memory-q-learning-for","title":"Convergence of Finite Memory Q-Learning for POMDPs and Near Optimality of Learned Policies under Filter Stability","date":"2021-03-22","arxiv_id":"2103.12158","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-based-on-scenario-tree","title":"Reinforcement Learning based on Scenario-tree MPC for ASVs","date":"2021-03-22","arxiv_id":"2103.11949","n_code_links":0,"syntology":null},{"paper":null,"slug":"variational-quantum-compiling-with-double-q","title":"Variational quantum compiling with double Q-learning","date":"2021-03-22","arxiv_id":"2103.11611","n_code_links":0,"syntology":null},{"paper":null,"slug":"full-gradient-dqn-reinforcement-learning-a","title":"Full Gradient DQN Reinforcement Learning: A Provably Convergent Scheme","date":"2021-03-10","arxiv_id":"2103.05981","n_code_links":0,"syntology":null},{"paper":null,"slug":"s4rl-surprisingly-simple-self-supervision-for","title":"S4RL: Surprisingly Simple Self-Supervision for Offline Reinforcement Learning","date":"2021-03-10","arxiv_id":"2103.06326","n_code_links":0,"syntology":null},{"paper":null,"slug":"increasing-energy-efficiency-of-massive-mimo","title":"Increasing Energy Efficiency of Massive-MIMO Network via Base Stations Switching using Reinforcement Learning and Radio Environment Maps","date":"2021-03-08","arxiv_id":"2103.11891","n_code_links":0,"syntology":null},{"paper":null,"slug":"correlated-deep-q-learning-based-microgrid","title":"Correlated Deep Q-learning based Microgrid Energy Management","date":"2021-03-06","arxiv_id":"2103.04152","n_code_links":0,"syntology":null},{"paper":null,"slug":"decentralized-microgrid-energy-management-a","title":"Decentralized Microgrid Energy Management: A Multi-agent Correlated Q-learning Approach","date":"2021-03-06","arxiv_id":"2103.04154","n_code_links":0,"syntology":null},{"paper":"/paper/super-resolving-compressed-images-via","slug":"super-resolving-compressed-images-via","title":"Super-resolving Compressed Images via Parallel and Series Integration of Artifact Reduction and Resolution Enhancement","date":"2021-03-02","arxiv_id":"2103.01698","n_code_links":1,"syntology":null},{"paper":"/paper/ucb-momentum-q-learning-correcting-the-bias","slug":"ucb-momentum-q-learning-correcting-the-bias","title":"UCB Momentum Q-learning: Correcting the bias without forgetting","date":"2021-03-01","arxiv_id":"2103.01312","n_code_links":1,"syntology":null},{"paper":null,"slug":"ensemble-bootstrapping-for-q-learning","title":"Ensemble Bootstrapping for Q-Learning","date":"2021-02-28","arxiv_id":"2103.00445","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-agent-path-planning-based-on-mpc-and","title":"Multi-Agent Path Planning based on MPC and DDPG","date":"2021-02-26","arxiv_id":"2102.13283","n_code_links":0,"syntology":null},{"paper":"/paper/balancing-rational-and-other-regarding","slug":"balancing-rational-and-other-regarding","title":"Balancing Rational and Other-Regarding Preferences in Cooperative-Competitive Environments","date":"2021-02-24","arxiv_id":"2102.12307","n_code_links":1,"syntology":null},{"paper":null,"slug":"sequential-learning-based-iaas-composition","title":"Sequential Learning-based IaaS Composition","date":"2021-02-24","arxiv_id":"2102.12598","n_code_links":0,"syntology":null},{"paper":null,"slug":"greedy-multi-step-off-policy-reinforcement-1","title":"Greedy-Step Off-Policy Reinforcement Learning","date":"2021-02-23","arxiv_id":"2102.11717","n_code_links":0,"syntology":null},{"paper":null,"slug":"stratified-experience-replay-correcting","title":"Stratified Experience Replay: Correcting Multiplicity Bias in Off-Policy Reinforcement Learning","date":"2021-02-22","arxiv_id":"2102.11319","n_code_links":0,"syntology":null},{"paper":"/paper/causal-inference-q-network-toward-resilient-1","slug":"causal-inference-q-network-toward-resilient-1","title":"Training a Resilient Q-Network against Observational Interference","date":"2021-02-18","arxiv_id":"2102.09677","n_code_links":1,"syntology":null},{"paper":"/paper/recurrent-rational-networks","slug":"recurrent-rational-networks","title":"Adaptive Rational Activations to Boost Deep Reinforcement Learning","date":"2021-02-18","arxiv_id":"2102.09407","n_code_links":4,"syntology":{"ran":6,"of":6,"n_ran_checked":0,"n_instrument":6,"unverified":0,"pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ml-research/rational_activations","ml-research/rational_rl","ml-research/rational_sl","k4ntz/activation-functions"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"finite-time-analysis-of-asynchronous-q","title":"A Discrete-Time Switching System Analysis of Q-learning","date":"2021-02-17","arxiv_id":"2102.08583","n_code_links":0,"syntology":null},{"paper":"/paper/dfac-framework-factorizing-the-value-function","slug":"dfac-framework-factorizing-the-value-function","title":"DFAC Framework: Factorizing the Value Function via Quantile Mixture for Multi-Agent Distributional Q-Learning","date":"2021-02-16","arxiv_id":"2102.07936","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["j3soon/dfac"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"cooperation-and-reputation-dynamics-with","title":"Cooperation and Reputation Dynamics with Reinforcement Learning","date":"2021-02-15","arxiv_id":"2102.07523","n_code_links":0,"syntology":null},{"paper":null,"slug":"reversible-action-design-for-combinatorial","title":"Reversible Action Design for Combinatorial Optimization with Reinforcement Learning","date":"2021-02-14","arxiv_id":"2102.07210","n_code_links":0,"syntology":null},{"paper":null,"slug":"tightening-the-dependence-on-horizon-in-the","title":"Is Q-Learning Minimax Optimal? A Tight Sample Complexity Analysis","date":"2021-02-12","arxiv_id":"2102.06548","n_code_links":0,"syntology":null},{"paper":"/paper/deep-reinforcement-agent-for-scheduling-in","slug":"deep-reinforcement-agent-for-scheduling-in","title":"Deep Reinforcement Agent for Scheduling in HPC","date":"2021-02-11","arxiv_id":"2102.06243","n_code_links":1,"syntology":null},{"paper":"/paper/benchmarking-deep-graph-generative-models-for","slug":"benchmarking-deep-graph-generative-models-for","title":"Benchmarking Deep Graph Generative Models for Optimizing New Drug Molecules for COVID-19","date":"2021-02-09","arxiv_id":"2102.04977","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-survey-of-motion-planning-algorithms-for","title":"A review of motion planning algorithms for intelligent robotics","date":"2021-02-04","arxiv_id":"2102.02376","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-a-reinforcement-learning-de-novo","title":"A step toward a reinforcement learning de novo genome assembler","date":"2021-02-02","arxiv_id":"2102.02649","n_code_links":0,"syntology":null},{"paper":"/paper/acting-in-delayed-environments-with-non-1","slug":"acting-in-delayed-environments-with-non-1","title":"Acting in Delayed Environments with Non-Stationary Markov Policies","date":"2021-01-28","arxiv_id":"2101.11992","n_code_links":2,"syntology":{"ran":3,"of":7,"n_ran_checked":1,"n_instrument":2,"unverified":4,"pointer_only":0,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","official":{"repos":["galdl/rl_delay_atari","galdl/rl_delay_basic"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"reinforcement-learning-based-per-antenna","title":"Reinforcement Learning based Per-antenna Discrete Power Control for Massive MIMO Systems","date":"2021-01-28","arxiv_id":"2101.12154","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-assisted-beamforming","title":"Reinforcement Learning Assisted Beamforming for Inter-cell Interference Mitigation in 5G Massive MIMO Networks","date":"2021-01-27","arxiv_id":"2103.11782","n_code_links":0,"syntology":null},{"paper":null,"slug":"channel-estimation-via-successive-denoising","title":"Channel Estimation via Successive Denoising in MIMO OFDM Systems: A Reinforcement Learning Approach","date":"2021-01-25","arxiv_id":"2101.10300","n_code_links":0,"syntology":null},{"paper":null,"slug":"fire-threat-detection-from-videos-with-q","title":"Fire Threat Detection From Videos with Q-Rough Sets","date":"2021-01-21","arxiv_id":"2101.08459","n_code_links":0,"syntology":null},{"paper":"/paper/benchmarking-perturbation-based-saliency-maps","slug":"benchmarking-perturbation-based-saliency-maps","title":"Benchmarking Perturbation-based Saliency Maps for Explaining Atari Agents","date":"2021-01-18","arxiv_id":"2101.07312","n_code_links":1,"syntology":null},{"paper":"/paper/randomized-ensembled-double-q-learning-1","slug":"randomized-ensembled-double-q-learning-1","title":"Randomized Ensembled Double Q-Learning: Learning Fast Without a Model","date":"2021-01-15","arxiv_id":"2101.05982","n_code_links":6,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":3,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","official":{"repos":["watchernyu/REDQ"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/continuous-deep-q-learning-with-simulator-for","slug":"continuous-deep-q-learning-with-simulator-for","title":"Continuous Deep Q-Learning with Simulator for Stabilization of Uncertain Discrete-Time Systems","date":"2021-01-13","arxiv_id":"2101.05640","n_code_links":1,"syntology":null},{"paper":"/paper/action-priors-for-large-action-spaces-in","slug":"action-priors-for-large-action-spaces-in","title":"Action Priors for Large Action Spaces in Robotics","date":"2021-01-11","arxiv_id":"2101.04178","n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-augmented-index-policy-for-optimal","title":"Learning Augmented Index Policy for Optimal Service Placement at the Network Edge","date":"2021-01-10","arxiv_id":"2101.03641","n_code_links":0,"syntology":null},{"paper":null,"slug":"robust-and-scalable-routing-with-multi-agent","title":"Robust and Scalable Routing with Multi-Agent Deep Reinforcement Learning for MANETs","date":"2021-01-09","arxiv_id":"2101.03273","n_code_links":0,"syntology":null},{"paper":"/paper/evolving-reinforcement-learning-algorithms-1","slug":"evolving-reinforcement-learning-algorithms-1","title":"Evolving Reinforcement Learning Algorithms","date":"2021-01-08","arxiv_id":"2101.03958","n_code_links":5,"syntology":null},{"paper":"/paper/federated-intelligence-for-active-queue","slug":"federated-intelligence-for-active-queue","title":"Federated Intelligence for Active Queue Management in Inter-Domain Congestion","date":"2021-01-08","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"reinforced-imitative-graph-representation","title":"Reinforced Imitative Graph Representation Learning for Mobile User Profiling: An Adversarial Training Perspective","date":"2021-01-07","arxiv_id":"2101.02634","n_code_links":0,"syntology":null},{"paper":"/paper/reinforcement-learning-with-latent-flow-1","slug":"reinforcement-learning-with-latent-flow-1","title":"Reinforcement Learning with Latent Flow","date":"2021-01-06","arxiv_id":"2101.01857","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["WendyShang/flare","WendyShang/dqn_zoo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/a-novel-policy-for-pre-trained-deep","slug":"a-novel-policy-for-pre-trained-deep","title":"A novel policy for pre-trained Deep Reinforcement Learning for Speech Emotion Recognition","date":"2021-01-04","arxiv_id":"2101.00738","n_code_links":1,"syntology":null},{"paper":null,"slug":"addressing-distribution-shift-in-online","title":"Addressing Distribution Shift in Online Reinforcement Learning with Offline Datasets","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-based-anti","title":"Deep Reinforcement Learning-based Anti-jamming Power Allocation in a Two-cell NOMA Network","date":"2021-01-01","arxiv_id":"2101.00270","n_code_links":0,"syntology":null},{"paper":null,"slug":"double-q-learning-new-analysis-and-sharper","title":"Double Q-learning: New Analysis and Sharper Finite-time Bound","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-movement-strategies-for-moving","title":"Learning Movement Strategies for Moving Target Defense","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-to-search-for-fast-maximum-common","title":"Learning to Search for Fast Maximum Common Subgraph Detection","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"pac-bayesian-randomized-value-function-with","title":"PAC-Bayesian Randomized Value Function with Informative Prior","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"preventing-value-function-collapse-in","title":"Preventing Value Function Collapse in Ensemble Q-Learning by Maximizing Representation Diversity","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"uncertainty-weighted-offline-reinforcement","title":"Uncertainty Weighted Offline Reinforcement Learning","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"weighted-bellman-backups-for-improved-signal","title":"Weighted Bellman Backups for Improved Signal-to-Noise in Q-Updates","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"disentangled-planning-and-control-in-vision","title":"Disentangled Planning and Control in Vision Based Robotics via Reward Machines","date":"2020-12-28","arxiv_id":"2012.14464","n_code_links":0,"syntology":null},{"paper":"/paper/popo-pessimistic-offline-policy-optimization","slug":"popo-pessimistic-offline-policy-optimization","title":"POPO: Pessimistic Offline Policy Optimization","date":"2020-12-26","arxiv_id":"2012.13682","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-state-representation-dueling-network-for","title":"A State Representation Dueling Network for Deep Reinforcement Learning","date":"2020-12-24","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"assured-rl-reinforcement-learning-with-almost","title":"Assured RL: Reinforcement Learning with Almost Sure Constraints","date":"2020-12-24","arxiv_id":"2012.13036","n_code_links":0,"syntology":null},{"paper":null,"slug":"distributed-q-learning-with-state-tracking","title":"Distributed Q-Learning with State Tracking for Multi-agent Networked Control","date":"2020-12-22","arxiv_id":"2012.12383","n_code_links":0,"syntology":null},{"paper":null,"slug":"goal-reasoning-by-selecting-subgoals-with","title":"Goal Reasoning by Selecting Subgoals with Deep Q-Learning","date":"2020-12-22","arxiv_id":"2012.12335","n_code_links":0,"syntology":null},{"paper":null,"slug":"deploying-reinforcement-learning-in-water","title":"Deploying Reinforcement Learning in Water Transport","date":"2020-12-14","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"learn-to-play-tetris-with-deep-reinforcement","title":"Learn to Play Tetris with Deep Reinforcement Learning","date":"2020-12-14","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"mobile-robots-autonomous-exploration-with","title":"Mobile Robots Autonomous Exploration with Reinforcement Learning","date":"2020-12-14","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/policy-gradient-rl-algorithms-as-directed","slug":"policy-gradient-rl-algorithms-as-directed","title":"Policy Gradient RL Algorithms as Directed Acyclic Graphs","date":"2020-12-14","arxiv_id":"2012.07763","n_code_links":1,"syntology":null},{"paper":null,"slug":"semi-supervised-off-policy-reinforcement","title":"Semi-Supervised Off Policy Reinforcement Learning","date":"2020-12-09","arxiv_id":"2012.04809","n_code_links":0,"syntology":null},{"paper":null,"slug":"selective-pseudo-labeling-with-reinforcement","title":"Selective Pseudo-Labeling with Reinforcement Learning for Semi-Supervised Domain Adaptation","date":"2020-12-07","arxiv_id":"2012.03438","n_code_links":0,"syntology":null},{"paper":null,"slug":"hippocampal-representations-emerge-when-1","title":"Hippocampal representations emerge when training recurrent neural networks on a memory dependent maze navigation task","date":"2020-12-02","arxiv_id":"2012.01328","n_code_links":0,"syntology":null},{"paper":null,"slug":"self-correcting-q-learning","title":"Self-correcting Q-Learning","date":"2020-12-02","arxiv_id":"2012.01100","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-new-convergent-variant-of-q-learning-with","title":"A new convergent variant of Q-learning with linear function approximation","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"a-unified-switching-system-perspective-and-1","title":"A Unified Switching System Perspective and Convergence Analysis of Q-Learning Algorithms","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"can-temporal-difference-and-q-learning-learn-1","title":"Can Temporal-Diﬀerence and Q-Learning Learn Representation? A Mean-Field Theory","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"robust-multi-agent-reinforcement-learning-1","title":"Robust Multi-Agent Reinforcement Learning with Model Uncertainty","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null}],"record_sha256":"392642b0945794c0a0f56fce8e1fc30c90dc32c27cbf153f696e0d377b5f7d60","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}