{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/109","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":109,"pages_in_order":152,"rows_per_page":100,"rows":[10801,10900],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/108","next":"/task/reinforcement-learning-1/papers/110","papers":[{"url":null,"slug":"representation-matters-offline-pretraining","title":"Representation Matters: Offline Pretraining for Sequential Decision Making","date":"2021-02-11","arxiv_id":"2102.05815","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-symmetric","title":"Deep Reinforcement Learning with Symmetric Prior for Predictive Power Allocation to Mobile Users","date":"2021-02-10","arxiv_id":"2103.13298","repositories_listed":0,"syntology":null},{"url":null,"slug":"defense-against-reward-poisoning-attacks-in","title":"Defense Against Reward Poisoning Attacks in Reinforcement Learning","date":"2021-02-10","arxiv_id":"2102.05776","repositories_listed":0,"syntology":null},{"url":null,"slug":"derivative-free-reinforcement-learning-a","title":"Derivative-Free Reinforcement Learning: A Review","date":"2021-02-10","arxiv_id":"2102.05710","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-equational-theorem-proving","title":"Learning Equational Theorem Proving","date":"2021-02-10","arxiv_id":"2102.05547","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-reinforcement-learning-for","title":"Leveraging Reinforcement Learning for evaluating Robustness of KNN Search Algorithms","date":"2021-02-10","arxiv_id":"2102.06525","repositories_listed":0,"syntology":null},{"url":null,"slug":"modeling-the-interaction-between-agents-in","title":"Modeling the Interaction between Agents in Cooperative Multi-Agent Reinforcement Learning","date":"2021-02-10","arxiv_id":"2102.06042","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-stationary-reinforcement-learning-without","title":"Non-stationary Reinforcement Learning without Prior Knowledge: An Optimal Black-box Approach","date":"2021-02-10","arxiv_id":"2102.05406","repositories_listed":0,"syntology":null},{"url":null,"slug":"patterns-predictions-and-actions-a-story","title":"Patterns, predictions, and actions: A story about machine learning","date":"2021-02-10","arxiv_id":"2102.05242","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalization-for-web-based-services-using","title":"Personalization for Web-based Services using Offline Reinforcement Learning","date":"2021-02-10","arxiv_id":"2102.05612","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-optimized-beam","title":"Reinforcement Learning for Optimized Beam Training in Multi-Hop Terahertz Communications","date":"2021-02-10","arxiv_id":"2102.05269","repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-averse-bayes-adaptive-reinforcement","title":"Risk-Averse Bayes-Adaptive Reinforcement Learning","date":"2021-02-10","arxiv_id":"2102.05762","repositories_listed":0,"syntology":null},{"url":null,"slug":"simple-agent-complex-environment-efficient","title":"Simple Agent, Complex Environment: Efficient Reinforcement Learning with Agent States","date":"2021-02-10","arxiv_id":"2102.05261","repositories_listed":0,"syntology":null},{"url":null,"slug":"measuring-progress-in-deep-reinforcement-1","title":"Measuring Progress in Deep Reinforcement Learning Sample Efficiency","date":"2021-02-09","arxiv_id":"2102.04881","repositories_listed":0,"syntology":null},{"url":null,"slug":"pairwise-weights-for-temporal-credit","title":"Adaptive Pairwise Weights for Temporal Credit Assignment","date":"2021-02-09","arxiv_id":"2102.04999","repositories_listed":0,"syntology":null},{"url":null,"slug":"scheduling-the-nasa-deep-space-network-with","title":"Scheduling the NASA Deep Space Network with Deep Reinforcement Learning","date":"2021-02-09","arxiv_id":"2102.05167","repositories_listed":0,"syntology":null},{"url":null,"slug":"contrasting-centralized-and-decentralized","title":"Contrasting Centralized and Decentralized Critics in Multi-Agent Reinforcement Learning","date":"2021-02-08","arxiv_id":"2102.04402","repositories_listed":0,"syntology":null},{"url":null,"slug":"generate-and-revise-reinforcement-learning-in","title":"Generate and Revise: Reinforcement Learning in Neural Poetry","date":"2021-02-08","arxiv_id":"2102.04114","repositories_listed":0,"syntology":null},{"url":null,"slug":"introduction-to-machine-learning-for-the","title":"Introduction to Machine Learning for the Sciences","date":"2021-02-08","arxiv_id":"2102.04883","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-optimal-strategies-for-temporal","title":"Learning Optimal Strategies for Temporal Tasks in Stochastic Games","date":"2021-02-08","arxiv_id":"2102.04307","repositories_listed":0,"syntology":null},{"url":null,"slug":"provable-model-based-nonlinear-bandit-and","title":"Provable Model-based Nonlinear Bandit and Reinforcement Learning: Shelve Optimism, Embrace Virtual Curvature","date":"2021-02-08","arxiv_id":"2102.04168","repositories_listed":0,"syntology":null},{"url":null,"slug":"unlocking-pixels-for-reinforcement-learning","title":"Unlocking Pixels for Reinforcement Learning via Implicit Attention","date":"2021-02-08","arxiv_id":"2102.04353","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-analysis-of-frame-skipping-in","title":"An Analysis of Frame-skipping in Reinforcement Learning","date":"2021-02-07","arxiv_id":"2102.03718","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-bandit-approach-to-curriculum-generation","title":"A bandit approach to curriculum generation for automatic speech recognition","date":"2021-02-06","arxiv_id":"2102.03662","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-hybrid-approach-for-reinforcement-learning","title":"A Hybrid Approach for Reinforcement Learning Using Virtual Policy Gradient for Balancing an Inverted Pendulum","date":"2021-02-06","arxiv_id":"2102.08362","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-modularized-and-scalable-multi-agent","title":"MSPM: A Modularized and Scalable Multi-Agent Reinforcement Learning-based System for Financial Portfolio Management","date":"2021-02-06","arxiv_id":"2102.03502","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-model-and-search-for-computer-go","title":"Improving Model and Search for Computer Go","date":"2021-02-06","arxiv_id":"2102.03467","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-deep-reinforcement-learning-for-7","title":"Multi-Agent Deep Reinforcement Learning for Request Dispatching in Distributed-Controller Software-Defined Networking","date":"2021-02-06","arxiv_id":"2103.03022","repositories_listed":0,"syntology":null},{"url":null,"slug":"addressing-inherent-uncertainty-risk","title":"Addressing Inherent Uncertainty: Risk-Sensitive Behavior Generation for Automated Driving using Distributional Reinforcement Learning","date":"2021-02-05","arxiv_id":"2102.03119","repositories_listed":0,"syntology":null},{"url":null,"slug":"deceptive-reinforcement-learning-for-privacy","title":"Deceptive Reinforcement Learning for Privacy-Preserving Planning","date":"2021-02-05","arxiv_id":"2102.03022","repositories_listed":0,"syntology":null},{"url":null,"slug":"experience-based-heuristic-search-robust","title":"Experience-Based Heuristic Search: Robust Motion Planning with Deep Q-Learning","date":"2021-02-05","arxiv_id":"2102.03127","repositories_listed":0,"syntology":null},{"url":null,"slug":"finite-sample-analysis-of-minimax-offline","title":"Finite Sample Analysis of Minimax Offline Reinforcement Learning: Completeness, Fast Rates and First-Order Efficiency","date":"2021-02-05","arxiv_id":"2102.02981","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-efficient-algorithms-for-multi","title":"Provably Efficient Algorithms for Multi-Objective Competitive RL","date":"2021-02-05","arxiv_id":"2102.03192","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-motion-planning-algorithms-for","title":"A review of motion planning algorithms for intelligent robotics","date":"2021-02-04","arxiv_id":"2102.02376","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-image-1","title":"Deep reinforcement learning-based image classification achieves perfect testing set accuracy for MRI brain tumors with a training set of only 30 images","date":"2021-02-04","arxiv_id":"2102.02895","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-to-train-your-robot-with-deep","title":"How to Train Your Robot with Deep Reinforcement Learning; Lessons We've Learned","date":"2021-02-04","arxiv_id":"2102.02915","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybrid-adversarial-inverse-reinforcement","title":"Hybrid Adversarial Imitation Learning","date":"2021-02-04","arxiv_id":"2102.02454","repositories_listed":0,"syntology":null},{"url":null,"slug":"persistent-rule-based-interactive","title":"Persistent Rule-based Interactive Reinforcement Learning","date":"2021-02-04","arxiv_id":"2102.02441","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-learning-model-for-gas-storage","title":"A deep learning model for gas storage optimization","date":"2021-02-03","arxiv_id":"2102.01980","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-a-compact-state-representation-for","title":"The Pitfall of More Powerful Autoencoders in Lidar-Based Navigation","date":"2021-02-03","arxiv_id":"2102.02127","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-uav-mobile-edge-computing-and-path","title":"Multi-UAV Mobile Edge Computing and Path Planning Platform based on Reinforcement Learning","date":"2021-02-03","arxiv_id":"2102.02078","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-recursive-belief-states-in-multi-agent","title":"Neural Recursive Belief States in Multi-Agent Reinforcement Learning","date":"2021-02-03","arxiv_id":"2102.02274","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-lyapunov-theory-for-finite-sample","title":"A Lyapunov Theory for Finite-Sample Guarantees of Asynchronous Q-Learning and TD-Learning Variants","date":"2021-02-02","arxiv_id":"2102.01567","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-abstraction-based-method-to-verify-multi","title":"An Abstraction-based Method to Check Multi-Agent Deep Reinforcement-Learning Behaviors","date":"2021-02-02","arxiv_id":"2102.01434","repositories_listed":0,"syntology":null},{"url":null,"slug":"approximately-solving-mean-field-games-via","title":"Approximately Solving Mean Field Games via Entropy-Regularized Deep Reinforcement Learning","date":"2021-02-02","arxiv_id":"2102.01585","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-reinforcement-learning-with-human","title":"Improving Reinforcement Learning with Human Assistance: An Argument for Human Subject Studies with HIPPO Gym","date":"2021-02-02","arxiv_id":"2102.02639","repositories_listed":0,"syntology":null},{"url":null,"slug":"near-optimal-offline-reinforcement-learning","title":"Near-Optimal Offline Reinforcement Learning via Double Variance Reduction","date":"2021-02-02","arxiv_id":"2102.01748","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-probabilistic-1","title":"Reinforcement Learning with Probabilistic Boolean Network Models of Smart Grid Devices","date":"2021-02-02","arxiv_id":"2102.01297","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-a-reinforcement-learning-de-novo","title":"A step toward a reinforcement learning de novo genome assembler","date":"2021-02-02","arxiv_id":"2102.02649","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-multi-agent-reinforcement-learning","title":"Towards Multi-agent Reinforcement Learning for Wireless Network Protocol Synthesis","date":"2021-02-02","arxiv_id":"2102.01611","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-secure-learning-control-strategy-via","title":"A Secure Learning Control Strategy via Dynamic Camouflaging for Unknown Dynamical Systems under Attacks","date":"2021-02-01","arxiv_id":"2102.00573","repositories_listed":0,"syntology":null},{"url":null,"slug":"bellman-eluder-dimension-new-rich-classes-of","title":"Bellman Eluder Dimension: New Rich Classes of RL Problems, and Sample-Efficient Algorithms","date":"2021-02-01","arxiv_id":"2102.00815","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybrid-beamforming-for-mmwave-mu-miso-systems","title":"Hybrid Beamforming for mmWave MU-MISO Systems Exploiting Multi-agent Deep Reinforcement Learning","date":"2021-02-01","arxiv_id":"2102.00735","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybrid-information-driven-multi-agent","title":"Hybrid Information-driven Multi-agent Reinforcement Learning","date":"2021-02-01","arxiv_id":"2102.01004","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-reinforcement-learning-inspired","title":"Interpretable Reinforcement Learning Inspired by Piaget's Theory of Cognitive Development","date":"2021-02-01","arxiv_id":"2102.00572","repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-aware-and-multi-objective-decision","title":"Risk Aware and Multi-Objective Decision Making with Distributional Monte Carlo Tree Search","date":"2021-02-01","arxiv_id":"2102.00966","repositories_listed":0,"syntology":null},{"url":null,"slug":"throughput-optimization-for-grant-free","title":"Throughput Optimization for Grant-Free Multiple Access With Multiagent Deep Reinforcement Learning","date":"2021-02-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-rates-for-the-regret-of-offline","title":"Fast Rates for the Regret of Offline Reinforcement Learning","date":"2021-01-31","arxiv_id":"2102.00479","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-human-decision-making-by","title":"Improving Human Decision-Making by Discovering Efficient Strategies for Hierarchical Planning","date":"2021-01-31","arxiv_id":"2102.00521","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-aided-monte-carlo","title":"Deep Reinforcement Learning Aided Monte Carlo Tree Search for MIMO Detection","date":"2021-01-30","arxiv_id":"2102.00178","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-product","title":"Deep Reinforcement Learning-Based Product Recommender for Online Advertising","date":"2021-01-30","arxiv_id":"2102.00333","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-stability-of-random-matrix-product","title":"On the Stability of Random Matrix Product with Markovian Noise: Application to Linear Stochastic Approximation and TD Learning","date":"2021-01-30","arxiv_id":"2102.00185","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-mirror-descent-for-reinforcement","title":"Policy Mirror Descent for Reinforcement Learning: Linear Convergence, New Sampling Complexity, and Generalized Problem Classes","date":"2021-01-30","arxiv_id":"2102.00135","repositories_listed":0,"syntology":null},{"url":null,"slug":"stay-alive-with-many-options-a-reinforcement","title":"Learning Skills to Navigate without a Master: A Sequential Multi-Policy Reinforcement Learning Algorithm","date":"2021-01-30","arxiv_id":"2102.00168","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-machine-learning-help-in-solving-cargo","title":"Reinforcement Learning for Freight Booking Control Problems","date":"2021-01-29","arxiv_id":"2102.00092","repositories_listed":0,"syntology":null},{"url":null,"slug":"challenges-for-using-impact-regularizers-to","title":"Challenges for Using Impact Regularizers to Avoid Negative Side Effects","date":"2021-01-29","arxiv_id":"2101.12509","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-based-vs-model-free-adaptive-control","title":"Learning-based vs Model-free Adaptive Control of a MAV under Wind Gust","date":"2021-01-29","arxiv_id":"2101.12501","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-voltage-control-using-structure","title":"Scalable Voltage Control using Structure-Driven Hierarchical Deep Reinforcement Learning","date":"2021-01-29","arxiv_id":"2102.00077","repositories_listed":0,"syntology":null},{"url":null,"slug":"thermal-control-of-laser-powder-bed-fusion","title":"Thermal Control of Laser Powder Bed Fusion Using Deep Reinforcement Learning","date":"2021-01-29","arxiv_id":"2102.03355","repositories_listed":0,"syntology":null},{"url":null,"slug":"coordiq-coordinated-q-learning-for-electric","title":"CoordiQ : Coordinated Q-learning for Electric Vehicle Charging Recommendation","date":"2021-01-28","arxiv_id":"2102.00847","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-per-antenna","title":"Reinforcement Learning based Per-antenna Discrete Power Control for Massive MIMO Systems","date":"2021-01-28","arxiv_id":"2101.12154","repositories_listed":0,"syntology":null},{"url":null,"slug":"universal-trading-for-order-execution-with","title":"Universal Trading for Order Execution with Oracle Policy Distillation","date":"2021-01-28","arxiv_id":"2103.10860","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-assisted-beamforming","title":"Reinforcement Learning Assisted Beamforming for Inter-cell Interference Mitigation in 5G Massive MIMO Networks","date":"2021-01-27","arxiv_id":"2103.11782","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-decision-making","title":"Reinforcement Learning for Selective Key Applications in Power Systems: Recent Advances and Future Challenges","date":"2021-01-27","arxiv_id":"2102.01168","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-android-malware-detection-system","title":"Robust Android Malware Detection System against Adversarial Attacks using Q-Learning","date":"2021-01-27","arxiv_id":"2101.12031","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-multi-agent-reinforcement-learning-via","title":"Safe Multi-Agent Reinforcement Learning via Shielding","date":"2021-01-27","arxiv_id":"2101.11196","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-minerl-2020-competition-on-sample","title":"The MineRL 2020 Competition on Sample Efficient Reinforcement Learning using Human Priors","date":"2021-01-26","arxiv_id":"2101.11071","repositories_listed":0,"syntology":null},{"url":null,"slug":"channel-estimation-via-successive-denoising","title":"Channel Estimation via Successive Denoising in MIMO OFDM Systems: A Reinforcement Learning Approach","date":"2021-01-25","arxiv_id":"2101.10300","repositories_listed":0,"syntology":null},{"url":null,"slug":"ecol-r-encouraging-copying-in-novel-object","title":"ECOL-R: Encouraging Copying in Novel Object Captioning with Reinforcement Learning","date":"2021-01-25","arxiv_id":"2101.09865","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-methodology-for-the-development-of-rl-based","title":"A Methodology for the Development of RL-Based Adaptive Traffic Signal Controllers","date":"2021-01-24","arxiv_id":"2101.09614","repositories_listed":0,"syntology":null},{"url":null,"slug":"episodic-memory-governs-choices-an-rnn-based","title":"Episodic memory governs choices: An RNN-based reinforcement learning model for decision-making task","date":"2021-01-24","arxiv_id":"2103.03679","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-sequence-generation-with-multi-agent","title":"Fast Sequence Generation with Multi-Agent Reinforcement Learning","date":"2021-01-24","arxiv_id":"2101.09698","repositories_listed":0,"syntology":null},{"url":null,"slug":"gst-group-sparse-training-for-accelerating","title":"GST: Group-Sparse Training for Accelerating Deep Reinforcement Learning","date":"2021-01-24","arxiv_id":"2101.09650","repositories_listed":0,"syntology":null},{"url":null,"slug":"solving-optimal-stopping-problems-with-deep-q","title":"Solving optimal stopping problems with Deep Q-Learning","date":"2021-01-24","arxiv_id":"2101.09682","repositories_listed":0,"syntology":null},{"url":null,"slug":"feature-selection-using-reinforcement","title":"Feature Selection Using Reinforcement Learning","date":"2021-01-23","arxiv_id":"2101.09460","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-exploration-for-sample-efficient","title":"Decoupled Exploration and Exploitation Policies for Sample-Efficient Reinforcement Learning","date":"2021-01-23","arxiv_id":"2101.09458","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-learning-and-optimization-techniques","title":"Safe Learning and Optimization Techniques: Towards a Survey of the State of the Art","date":"2021-01-23","arxiv_id":"2101.09505","repositories_listed":0,"syntology":null},{"url":null,"slug":"prior-preference-learning-from-experts-1","title":"Prior Preference Learning from Experts:Designing a Reward with Active Inference","date":"2021-01-22","arxiv_id":"2101.08937","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-machine-learning-for-flooding","title":"Adversarial Machine Learning for Flooding Attacks on 5G Radio Access Network Slicing","date":"2021-01-21","arxiv_id":"2101.08724","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-policy-search-for-partially","title":"Model-based Policy Search for Partially Measurable Systems","date":"2021-01-21","arxiv_id":"2101.08740","repositories_listed":0,"syntology":null},{"url":null,"slug":"collision-free-flocking-with-a-dynamic-squad","title":"Flocking and Collision Avoidance for a Dynamic Squad of Fixed-Wing UAVs Using Deep Reinforcement Learning","date":"2021-01-20","arxiv_id":"2101.08074","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-optimizes","title":"Deep Reinforcement Learning Optimizes Graphene Nanopores for Efficient Desalination","date":"2021-01-19","arxiv_id":"2101.07399","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-bicycle-dispatching-of-dockless","title":"Dynamic Bicycle Dispatching of Dockless Public Bicycle-sharing Systems using Multi-objective Reinforcement Learning","date":"2021-01-19","arxiv_id":"2101.07437","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-for-adaptive","title":"Meta-Reinforcement Learning for Adaptive Motor Control in Changing Robot Dynamics and Environments","date":"2021-01-19","arxiv_id":"2101.07599","repositories_listed":0,"syntology":null},{"url":null,"slug":"spatial-assembly-generative-architecture-with","title":"Spatial Assembly: Generative Architecture With Reinforcement Learning, Self Play and Tree Search","date":"2021-01-19","arxiv_id":"2101.07579","repositories_listed":0,"syntology":null},{"url":"/paper/claster-clustering-with-reinforcement","slug":"claster-clustering-with-reinforcement","title":"CLASTER: Clustering with Reinforcement Learning for Zero-Shot Action Recognition","date":"2021-01-18","arxiv_id":"2101.07042","repositories_listed":0,"syntology":null},{"url":null,"slug":"cooperative-and-competitive-biases-for-multi","title":"Cooperative and Competitive Biases for Multi-Agent Reinforcement Learning","date":"2021-01-18","arxiv_id":"2101.06890","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-embedded-lqr","title":"Deep Reinforcement Learning with Embedded LQR Controllers","date":"2021-01-18","arxiv_id":"2101.07175","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-reinforcement-learning-for-3","title":"Model-Based Reinforcement Learning for Approximate Optimal Control with Temporal Logic Specifications","date":"2021-01-18","arxiv_id":"2101.07156","repositories_listed":0,"syntology":null},{"url":null,"slug":"regularized-policies-are-reward-robust","title":"Regularized Policies are Reward Robust","date":"2021-01-18","arxiv_id":"2101.07012","repositories_listed":0,"syntology":null}],"record_sha256":"c4944ad82f7d3cf5e3a2d1ec293f79d4665bc345cda0ec812d5b075a63b0e70d","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}