{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/q-learning/papers/14","list_of":"/method/q-learning","method":"Q-Learning","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":14,"pages_in_order":18,"rows_per_page":100,"rows":[1301,1400],"of":1734,"counts":{"archive_papers_tagged":1734,"with_a_code_link":464,"where_syntology_ran_a_sample":126,"not_listed_spam_title":0,"listed":1734,"listed_where_code_ran":126,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":105,"every_run_a_failure_of_syntologys_instrument":21,"listed_with_a_run_with_no_instrument_failure":105,"listed_every_run_a_failure_of_syntologys_instrument":21,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/q-learning","prev":"/method/q-learning/papers/13","next":"/method/q-learning/papers/15","papers":[{"paper":null,"slug":"periodic-q-learning","title":"Periodic Q-Learning","date":"2020-02-23","arxiv_id":"2002.09795","n_code_links":0,"syntology":null},{"paper":null,"slug":"anypath-routing-protocol-design-via-q","title":"Anypath Routing Protocol Design via Q-Learning for Underwater Sensor Networks","date":"2020-02-22","arxiv_id":"2002.09623","n_code_links":0,"syntology":null},{"paper":null,"slug":"disentangling-controllable-object-through","title":"Disentangling Controllable Object through Video Prediction Improves Visual Reinforcement Learning","date":"2020-02-21","arxiv_id":"2002.09136","n_code_links":0,"syntology":null},{"paper":"/paper/langevin-dqn","slug":"langevin-dqn","title":"Langevin DQN","date":"2020-02-17","arxiv_id":"2002.07282","n_code_links":2,"syntology":null},{"paper":null,"slug":"a-multimodal-dialogue-system-for","title":"A Multimodal Dialogue System for Conversational Image Editing","date":"2020-02-16","arxiv_id":"2002.06484","n_code_links":0,"syntology":null},{"paper":"/paper/maxmin-q-learning-controlling-the-estimation-1","slug":"maxmin-q-learning-controlling-the-estimation-1","title":"Maxmin Q-learning: Controlling the Estimation Bias of Q-learning","date":"2020-02-16","arxiv_id":"2002.06487","n_code_links":1,"syntology":null},{"paper":"/paper/reinforced-active-learning-for-image-1","slug":"reinforced-active-learning-for-image-1","title":"Reinforced active learning for image segmentation","date":"2020-02-16","arxiv_id":"2002.06583","n_code_links":1,"syntology":null},{"paper":null,"slug":"fast-reinforcement-learning-for-anti-jamming","title":"Fast Reinforcement Learning for Anti-jamming Communications","date":"2020-02-13","arxiv_id":"2002.05364","n_code_links":0,"syntology":null},{"paper":null,"slug":"listwise-learning-to-rank-with-deep-q","title":"Listwise Learning to Rank with Deep Q-Networks","date":"2020-02-13","arxiv_id":"2002.07651","n_code_links":0,"syntology":null},{"paper":null,"slug":"regret-bounds-for-discounted-mdps","title":"Regret Bounds for Discounted MDPs","date":"2020-02-12","arxiv_id":"2002.05138","n_code_links":0,"syntology":null},{"paper":null,"slug":"q-learning-for-mean-field-controls","title":"Mean-Field Controls with Q-learning for Cooperative MARL: Convergence and Complexity Analysis","date":"2020-02-10","arxiv_id":"2002.04131","n_code_links":0,"syntology":null},{"paper":"/paper/learning-state-abstractions-for-transfer-in","slug":"learning-state-abstractions-for-transfer-in","title":"Learning State Abstractions for Transfer in Continuous Control","date":"2020-02-08","arxiv_id":"2002.05518","n_code_links":2,"syntology":null},{"paper":null,"slug":"safe-wasserstein-constrained-deep-q-learning","title":"Safe Wasserstein Constrained Deep Q-Learning","date":"2020-02-07","arxiv_id":"2002.03016","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-rbf-value-functions-for-continuous","title":"Deep Radial-Basis Value Functions for Continuous Control","date":"2020-02-05","arxiv_id":"2002.01883","n_code_links":0,"syntology":null},{"paper":"/paper/interpretable-end-to-end-urban-autonomous","slug":"interpretable-end-to-end-urban-autonomous","title":"Interpretable End-to-end Urban Autonomous Driving with Latent Deep Reinforcement Learning","date":"2020-01-23","arxiv_id":"2001.08726","n_code_links":4,"syntology":{"ran":2,"of":6,"n_ran_checked":2,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["cjy1992/interp-e2e-driving"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"q-learning-in-enormous-action-spaces-via","title":"Q-Learning in enormous action spaces via amortized approximate maximization","date":"2020-01-22","arxiv_id":"2001.08116","n_code_links":0,"syntology":null},{"paper":"/paper/discriminator-soft-actor-critic-without","slug":"discriminator-soft-actor-critic-without","title":"Discriminator Soft Actor Critic without Extrinsic Rewards","date":"2020-01-19","arxiv_id":"2001.06808","n_code_links":1,"syntology":null},{"paper":null,"slug":"model-based-multi-agent-reinforcement","title":"Model-based Multi-Agent Reinforcement Learning with Cooperative Prioritized Sweeping","date":"2020-01-15","arxiv_id":"2001.07527","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-interactive-reinforcement-learning-for","title":"Deep Interactive Reinforcement Learning for Path Following of Autonomous Underwater Vehicle","date":"2020-01-10","arxiv_id":"2001.03359","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-probabilistic-simulator-of-spatial-demand","title":"A Probabilistic Simulator of Spatial Demand for Product Allocation","date":"2020-01-09","arxiv_id":"2001.03210","n_code_links":0,"syntology":null},{"paper":null,"slug":"eeg-based-drowsiness-estimation-for-driving","title":"EEG-based Drowsiness Estimation for Driving Safety using Deep Q-Learning","date":"2020-01-08","arxiv_id":"2001.02399","n_code_links":0,"syntology":null},{"paper":null,"slug":"experimental-analysis-of-reinforcement","title":"Experimental Analysis of Reinforcement Learning Techniques for Spectrum Sharing Radar","date":"2020-01-06","arxiv_id":"2001.01799","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-randomized-least-squares-value-iteration","title":"Deep Randomized Least Squares Value Iteration","date":"2020-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"svqn-sequential-variational-soft-q-learning","title":"SVQN: Sequential Variational Soft Q-Learning Networks","date":"2020-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"way-off-policy-batch-deep-reinforcement-1","title":"Way Off-Policy Batch Deep Reinforcement Learning of Human Preferences in Dialog","date":"2020-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"information-theoretic-model-predictive-q-1","title":"Information Theoretic Model Predictive Q-Learning","date":"2019-12-31","arxiv_id":"2001.02153","n_code_links":0,"syntology":null},{"paper":"/paper/slm-lab-a-comprehensive-benchmark-and-modular-1","slug":"slm-lab-a-comprehensive-benchmark-and-modular-1","title":"SLM Lab: A Comprehensive Benchmark and Modular Software Framework for Reproducible Deep Reinforcement Learning","date":"2019-12-28","arxiv_id":"1912.12482","n_code_links":1,"syntology":{"ran":10,"of":13,"n_ran_checked":10,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["kengz/SLM-Lab"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"hamilton-jacobi-bellman-equations-for-q","title":"Hamilton-Jacobi-Bellman Equations for Q-Learning in Continuous Time","date":"2019-12-23","arxiv_id":"1912.10697","n_code_links":0,"syntology":null},{"paper":"/paper/learning-an-interpretable-traffic-signal","slug":"learning-an-interpretable-traffic-signal","title":"Learning an Interpretable Traffic Signal Control Policy","date":"2019-12-23","arxiv_id":"1912.11023","n_code_links":1,"syntology":null},{"paper":null,"slug":"exploiting-the-potential-of-deep","title":"Exploiting the potential of deep reinforcement learning for classification tasks in high-dimensional and unstructured data","date":"2019-12-20","arxiv_id":"1912.09595","n_code_links":0,"syntology":null},{"paper":null,"slug":"soft-q-network","title":"Soft Q Network","date":"2019-12-20","arxiv_id":"1912.10891","n_code_links":0,"syntology":null},{"paper":null,"slug":"sepsis-world-model-a-mimic-based-openai-gym","title":"Sepsis World Model: A MIMIC-based OpenAI Gym \"World Model\" Simulator for Sepsis Treatment","date":"2019-12-15","arxiv_id":"1912.07127","n_code_links":0,"syntology":null},{"paper":null,"slug":"high-dimensional-precision-medicine-from","title":"High dimensional precision medicine from patient-derived xenografts","date":"2019-12-13","arxiv_id":"1912.06667","n_code_links":0,"syntology":null},{"paper":null,"slug":"provably-efficient-reinforcement-learning-1","title":"Provably Efficient Reinforcement Learning with Aggregated States","date":"2019-12-13","arxiv_id":"1912.06366","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-finite-time-analysis-of-q-learning-with-1","title":"A Finite-Time Analysis of Q-Learning with Neural Network Function Approximation","date":"2019-12-10","arxiv_id":"1912.04511","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-sparse-representations-incrementally","title":"Learning Sparse Representations Incrementally in Deep Reinforcement Learning","date":"2019-12-09","arxiv_id":"1912.04002","n_code_links":0,"syntology":null},{"paper":null,"slug":"value-of-information-based-arbitration","title":"Value-of-Information based Arbitration between Model-based and Model-free Control","date":"2019-12-08","arxiv_id":"1912.05453","n_code_links":0,"syntology":null},{"paper":null,"slug":"191202552","title":"Reinforcement Learning with Non-Markovian Rewards","date":"2019-12-05","arxiv_id":"1912.02552","n_code_links":0,"syntology":null},{"paper":null,"slug":"combining-q-learning-and-search-with-1","title":"Combining Q-Learning and Search with Amortized Value Estimates","date":"2019-12-05","arxiv_id":"1912.02807","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-unified-switching-system-perspective-and","title":"A Unified Switching System Perspective and O.D.E. Analysis of Q-Learning Algorithms","date":"2019-12-04","arxiv_id":"1912.02270","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-to-dynamically-coordinate-multi","title":"Learning to Dynamically Coordinate Multi-Robot Teams in Graph Attention Networks","date":"2019-12-04","arxiv_id":"1912.02059","n_code_links":0,"syntology":null},{"paper":null,"slug":"neighborhood-cognition-consistent-multi-agent","title":"Neighborhood Cognition Consistent Multi-Agent Reinforcement Learning","date":"2019-12-03","arxiv_id":"1912.01160","n_code_links":0,"syntology":null},{"paper":null,"slug":"modelling-the-dynamics-of-multiagent-q","title":"Modelling the Dynamics of Multiagent Q-Learning in Repeated Symmetric Games: a Mean Field Theoretic Approach","date":"2019-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/propagating-uncertainty-in-reinforcement","slug":"propagating-uncertainty-in-reinforcement","title":"Propagating Uncertainty in Reinforcement Learning via Wasserstein Barycenters","date":"2019-12-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"provably-efficient-q-learning-with-function-1","title":"Provably Efficient Q-learning with Function Approximation via Distribution Shift Error Checking Oracle","date":"2019-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/reconciling-returns-with-experience-replay","slug":"reconciling-returns-with-experience-replay","title":"Reconciling λ-Returns with Experience Replay","date":"2019-12-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"quadratic-q-network-for-learning-continuous","title":"Quadratic Q-network for Learning Continuous Control for Autonomous Vehicles","date":"2019-11-29","arxiv_id":"1912.00074","n_code_links":0,"syntology":null},{"paper":null,"slug":"control-tutored-reinforcement-learning-an","title":"Control-Tutored Reinforcement Learning: an application to the Herding Problem","date":"2019-11-26","arxiv_id":"1911.11444","n_code_links":0,"syntology":null},{"paper":"/paper/join-query-optimization-with-deep","slug":"join-query-optimization-with-deep","title":"Join Query Optimization with Deep Reinforcement Learning Algorithms","date":"2019-11-26","arxiv_id":"1911.11689","n_code_links":1,"syntology":null},{"paper":null,"slug":"adaptive-modulation-and-coding-based-on","title":"Adaptive Modulation and Coding based on Reinforcement Learning for 5G Networks","date":"2019-11-25","arxiv_id":"1912.04030","n_code_links":0,"syntology":null},{"paper":null,"slug":"mitigate-bias-in-face-recognition-using","title":"Mitigate Bias in Face Recognition using Skewness-Aware Reinforcement Learning","date":"2019-11-25","arxiv_id":"1911.10692","n_code_links":0,"syntology":null},{"paper":null,"slug":"which-channel-to-ask-my-question-personalized","title":"Which Channel to Ask My Question? Personalized Customer Service RequestStream Routing using DeepReinforcement Learning","date":"2019-11-24","arxiv_id":"1911.10521","n_code_links":0,"syntology":null},{"paper":null,"slug":"efficient-drone-mobility-support-using","title":"Efficient Drone Mobility Support Using Reinforcement Learning","date":"2019-11-21","arxiv_id":"1911.09715","n_code_links":0,"syntology":null},{"paper":null,"slug":"quantum-observables-for-continuous-control-of","title":"Quantum Observables for continuous control of the Quantum Approximate Optimization Algorithm via Reinforcement Learning","date":"2019-11-21","arxiv_id":"1911.09682","n_code_links":0,"syntology":null},{"paper":null,"slug":"placement-optimization-of-aerial-base","title":"Placement Optimization of Aerial Base Stations with Deep Reinforcement Learning","date":"2019-11-19","arxiv_id":"1911.08111","n_code_links":0,"syntology":null},{"paper":null,"slug":"asymptotics-of-reinforcement-learning-with","title":"Asymptotics of Reinforcement Learning with Neural Networks","date":"2019-11-13","arxiv_id":"1911.07304","n_code_links":0,"syntology":null},{"paper":null,"slug":"minimalistic-attacks-how-little-it-takes-to","title":"Minimalistic Attacks: How Little it Takes to Fool a Deep Reinforcement Learning Policy","date":"2019-11-10","arxiv_id":"1911.03849","n_code_links":0,"syntology":null},{"paper":null,"slug":"two-stage-wecc-composite-load-modeling-a","title":"Two-stage WECC Composite Load Modeling: A Double Deep Q-Learning Networks Approach","date":"2019-11-08","arxiv_id":"1911.04894","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-end-to-end-deep-rl-framework-for-task","title":"An End-to-End Deep RL Framework for Task Arrangement in Crowdsourcing Platforms","date":"2019-11-04","arxiv_id":"1911.01030","n_code_links":0,"syntology":null},{"paper":null,"slug":"challenging-on-car-racing-problem-from-openai","title":"Challenging On Car Racing Problem from OpenAI gym","date":"2019-11-02","arxiv_id":"1911.04868","n_code_links":0,"syntology":null},{"paper":"/paper/on-solving-the-2-dimensional-greedy-shooter","slug":"on-solving-the-2-dimensional-greedy-shooter","title":"On Solving the 2-Dimensional Greedy Shooter Problem for UAVs","date":"2019-11-02","arxiv_id":"1911.01419","n_code_links":1,"syntology":null},{"paper":"/paper/generalized-speedy-q-learning","slug":"generalized-speedy-q-learning","title":"Generalized Speedy Q-learning","date":"2019-11-01","arxiv_id":"1911.00397","n_code_links":1,"syntology":null},{"paper":null,"slug":"biomimetic-ultra-broadband-perfect-absorbers","title":"Biomimetic Ultra-Broadband Perfect Absorbers Optimised with Reinforcement Learning","date":"2019-10-28","arxiv_id":"1910.12465","n_code_links":0,"syntology":null},{"paper":null,"slug":"model-free-mean-field-reinforcement-learning","title":"Model-Free Mean-Field Reinforcement Learning: Mean-Field MDP and Mean-Field Q-Learning","date":"2019-10-28","arxiv_id":"1910.12802","n_code_links":0,"syntology":null},{"paper":"/paper/bail-best-action-imitation-learning-for-batch-1","slug":"bail-best-action-imitation-learning-for-batch-1","title":"BAIL: Best-Action Imitation Learning for Batch Deep Reinforcement Learning","date":"2019-10-27","arxiv_id":"1910.12179","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":1,"n_instrument":4,"unverified":1,"pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","official":{"repos":["lanyavik/BAIL"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/task-oriented-language-grounding-for-language","slug":"task-oriented-language-grounding-for-language","title":"Task-Oriented Language Grounding for Language Input with Multiple Sub-Goals of Non-Linear Order","date":"2019-10-27","arxiv_id":"1910.12354","n_code_links":1,"syntology":null},{"paper":null,"slug":"d-point-trigonometric-path-planning-based-on","title":"D-Point Trigonometric Path Planning based on Q-Learning in Uncertain Environments","date":"2019-10-26","arxiv_id":"1910.12020","n_code_links":0,"syntology":null},{"paper":"/paper/zpd-teaching-strategies-for-deep","slug":"zpd-teaching-strategies-for-deep","title":"ZPD Teaching Strategies for Deep Reinforcement Learning from Demonstrations","date":"2019-10-26","arxiv_id":"1910.12154","n_code_links":2,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"deep-q-learning-for-same-day-delivery-with-a","title":"Deep Q-Learning for Same-Day Delivery with Vehicles and Drones","date":"2019-10-25","arxiv_id":"1910.11901","n_code_links":0,"syntology":null},{"paper":null,"slug":"momentum-in-reinforcement-learning","title":"Momentum in Reinforcement Learning","date":"2019-10-21","arxiv_id":"1910.09322","n_code_links":0,"syntology":null},{"paper":null,"slug":"resource-allocation-in-mobility-aware","title":"Resource Allocation in Mobility-Aware Federated Learning Networks: A Deep Reinforcement Learning Approach","date":"2019-10-21","arxiv_id":"1910.09172","n_code_links":0,"syntology":null},{"paper":"/paper/policy-learning-for-malaria-control","slug":"policy-learning-for-malaria-control","title":"Policy Learning for Malaria Control","date":"2019-10-20","arxiv_id":"1910.08926","n_code_links":2,"syntology":null},{"paper":null,"slug":"reverse-experience-replay","title":"Reverse Experience Replay","date":"2019-10-19","arxiv_id":"1910.08780","n_code_links":0,"syntology":null},{"paper":"/paper/automatic-data-augmentation-by-learning-the","slug":"automatic-data-augmentation-by-learning-the","title":"Automatic Data Augmentation by Learning the Deterministic Policy","date":"2019-10-18","arxiv_id":"1910.08343","n_code_links":1,"syntology":null},{"paper":null,"slug":"on-the-reduction-of-variance-and","title":"On the Reduction of Variance and Overestimation of Deep Q-Learning","date":"2019-10-14","arxiv_id":"1910.05983","n_code_links":0,"syntology":null},{"paper":null,"slug":"zap-q-learning-with-nonlinear-function","title":"Zap Q-Learning With Nonlinear Function Approximation","date":"2019-10-11","arxiv_id":"1910.05405","n_code_links":0,"syntology":null},{"paper":null,"slug":"integrating-behavior-cloning-and","title":"Integrating Behavior Cloning and Reinforcement Learning for Improved Performance in Dense and Sparse Reward Environments","date":"2019-10-09","arxiv_id":"1910.04281","n_code_links":0,"syntology":null},{"paper":null,"slug":"knowledge-induced-deep-q-network-for-a-slide","title":"Learning Visual Affordances with Target-Orientated Deep Q-Network to Grasp Objects by Harnessing Environmental Fixtures","date":"2019-10-09","arxiv_id":"1910.03781","n_code_links":0,"syntology":null},{"paper":null,"slug":"toward-synergic-learning-for-autonomous","title":"Toward Synergic Learning for Autonomous Manipulation of Deformable Tissues via Surgical Robots: An Approximate Q-Learning Approach","date":"2019-10-08","arxiv_id":"1910.03398","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-step-greedy-policies-in-model-free-deep-1","title":"Multi-step Greedy Reinforcement Learning Algorithms","date":"2019-10-07","arxiv_id":"1910.02919","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-with-structured-1","title":"Reinforcement Learning with Structured Hierarchical Grammar Representations of Actions","date":"2019-10-07","arxiv_id":"1910.02876","n_code_links":0,"syntology":null},{"paper":"/paper/deep-q-network-for-angry-birds","slug":"deep-q-network-for-angry-birds","title":"Deep Q-Network for Angry Birds","date":"2019-10-04","arxiv_id":"1910.01806","n_code_links":1,"syntology":null},{"paper":null,"slug":"im-sorry-dave-im-afraid-i-cant-do-that-deep-q","title":"I'm sorry Dave, I'm afraid I can't do that, Deep Q-learning from forbidden action","date":"2019-10-04","arxiv_id":"1910.02078","n_code_links":0,"syntology":null},{"paper":"/paper/benchmarking-batch-deep-reinforcement","slug":"benchmarking-batch-deep-reinforcement","title":"Benchmarking Batch Deep Reinforcement Learning Algorithms","date":"2019-10-03","arxiv_id":"1910.01708","n_code_links":5,"syntology":null},{"paper":null,"slug":"ai-assisted-annotator-using-reinforcement","title":"AI Assisted Annotator using Reinforcement Learning","date":"2019-10-02","arxiv_id":"1910.02052","n_code_links":0,"syntology":null},{"paper":null,"slug":"quantile-qt-opt-for-risk-aware-vision-based","title":"Quantile QT-Opt for Risk-Aware Vision-Based Robotic Grasping","date":"2019-10-01","arxiv_id":"1910.02787","n_code_links":0,"syntology":null},{"paper":"/paper/meta-q-learning","slug":"meta-q-learning","title":"Meta-Q-Learning","date":"2019-09-30","arxiv_id":"1910.00125","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["amazon-research/meta-q-learning"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"off-policy-multi-step-q-learning","title":"Composite Q-learning: Multi-scale Q-function Decomposition and Separable Optimization","date":"2019-09-30","arxiv_id":"1909.13518","n_code_links":0,"syntology":null},{"paper":"/paper/a-simulation-of-uav-power-optimization-via","slug":"a-simulation-of-uav-power-optimization-via","title":"Visual Exploration and Energy-aware Path Planning via Reinforcement Learning","date":"2019-09-26","arxiv_id":"1909.12217","n_code_links":1,"syntology":null},{"paper":null,"slug":"caql-continuous-action-q-learning","title":"CAQL: Continuous Action Q-Learning","date":"2019-09-26","arxiv_id":"1909.12397","n_code_links":0,"syntology":null},{"paper":"/paper/demystifying-active-inference","slug":"demystifying-active-inference","title":"Active inference: demystified and compared","date":"2019-09-24","arxiv_id":"1909.10863","n_code_links":1,"syntology":null},{"paper":"/paper/190909902","slug":"190909902","title":"Deep Reinforcement Learning with Modulated Hebbian plus Q Network Architecture","date":"2019-09-21","arxiv_id":"1909.09902","n_code_links":1,"syntology":null},{"paper":null,"slug":"on-the-convergence-of-approximate-and","title":"On the Convergence of Approximate and Regularized Policy Iteration Schemes","date":"2019-09-20","arxiv_id":"1909.09621","n_code_links":0,"syntology":null},{"paper":"/paper/modelicagym-applying-reinforcement-learning","slug":"modelicagym-applying-reinforcement-learning","title":"ModelicaGym: Applying Reinforcement Learning to Modelica Models","date":"2019-09-18","arxiv_id":"1909.08604","n_code_links":1,"syntology":null},{"paper":null,"slug":"split-deep-q-learning-for-robust-object","title":"Split Deep Q-Learning for Robust Object Singulation","date":"2019-09-17","arxiv_id":"1909.08105","n_code_links":0,"syntology":null},{"paper":null,"slug":"joint-inference-of-reward-machines-and","title":"Joint Inference of Reward Machines and Policies for Reinforcement Learning","date":"2019-09-12","arxiv_id":"1909.05912","n_code_links":0,"syntology":null},{"paper":null,"slug":"mutual-information-regularization-in-markov","title":"Mutual-Information Regularization in Markov Decision Processes and Actor-Critic Learning","date":"2019-09-11","arxiv_id":"1909.05950","n_code_links":0,"syntology":null},{"paper":null,"slug":"q-learning-assisted-energy-aware-traffic","title":"Q-learning Assisted Energy-Aware Traffic Offloading and Cell Switching in Heterogeneous Networks","date":"2019-09-11","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"a-multistep-lyapunov-approach-for-finite-time","title":"A Multistep Lyapunov Approach for Finite-Time Analysis of Biased Stochastic Approximation","date":"2019-09-10","arxiv_id":"1909.04299","n_code_links":0,"syntology":null},{"paper":null,"slug":"fixed-horizon-temporal-difference-methods-for","title":"Fixed-Horizon Temporal Difference Methods for Stable Reinforcement Learning","date":"2019-09-09","arxiv_id":"1909.03906","n_code_links":0,"syntology":null}],"record_sha256":"a4a8f1e54d32dc942f24bf0b1ba470b24a4b50e61dc1712fb3d12c118a3dc144","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}