{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/111","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":111,"pages_in_order":135,"rows_per_page":100,"rows":[11001,11100],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/110","next":"/task/reinforcement-learning-2/papers/112","papers":[{"url":null,"slug":"distributed-reinforcement-learning-for-1","title":"Distributed Reinforcement Learning for Cooperative Multi-Robot Object Manipulation","date":"2020-03-21","arxiv_id":"2003.09540","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-weighted-q","title":"Deep Reinforcement Learning with Weighted Q-Learning","date":"2020-03-20","arxiv_id":"2003.09280","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-sets-for-generalization-in-rl","title":"Deep Sets for Generalization in RL","date":"2020-03-20","arxiv_id":"2003.09443","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-multi-time-scale-constraints-in","title":"Deep Constrained Q-learning","date":"2020-03-20","arxiv_id":"2003.09398","repositories_listed":0,"syntology":null},{"url":null,"slug":"exchangeable-input-representations-for","title":"Exchangeable Input Representations for Reinforcement Learning","date":"2020-03-19","arxiv_id":"2003.09022","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-enabled-cooperative","title":"Reinforcement learning enabled cooperative spectrum sensing in cognitive radio networks","date":"2020-03-19","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-cognitive-routing-based-on-deep","title":"Towards Cognitive Routing based on Deep Reinforcement Learning","date":"2020-03-19","arxiv_id":"2003.12439","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-socially-acceptable-perturbations","title":"Generating Socially Acceptable Perturbations for Efficient Evaluation of Autonomous Vehicles","date":"2020-03-18","arxiv_id":"2003.08034","repositories_listed":0,"syntology":null},{"url":null,"slug":"placement-optimization-with-deep","title":"Placement Optimization with Deep Reinforcement Learning","date":"2020-03-18","arxiv_id":"2003.08445","repositories_listed":0,"syntology":null},{"url":null,"slug":"viewport-aware-deep-reinforcement-learning","title":"Viewport-Aware Deep Reinforcement Learning Approach for 360$^o$ Video Caching","date":"2020-03-18","arxiv_id":"2003.08473","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-performance-in-reinforcement","title":"Improving Performance in Reinforcement Learning by Breaking Generalization in Neural Networks","date":"2020-03-16","arxiv_id":"2003.07417","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-electricity","title":"Reinforcement Learning for Electricity Network Operation","date":"2020-03-16","arxiv_id":"2003.07339","repositories_listed":0,"syntology":null},{"url":null,"slug":"cooperation-without-coordination-hierarchical","title":"Model-based Reinforcement Learning for Decentralized Multiagent Rendezvous","date":"2020-03-15","arxiv_id":"2003.06906","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-general-framework-for-learning-mean-field","title":"A General Framework for Learning Mean-Field Games","date":"2020-03-13","arxiv_id":"2003.06069","repositories_listed":0,"syntology":null},{"url":"/paper/taylor-expansion-policy-optimization","slug":"taylor-expansion-policy-optimization","title":"Taylor Expansion Policy Optimization","date":"2020-03-13","arxiv_id":"2003.06259","repositories_listed":0,"syntology":null},{"url":null,"slug":"heterogeneous-relational-reasoning-in","title":"Heterogeneous Relational Reasoning in Knowledge Graphs with Reinforcement Learning","date":"2020-03-12","arxiv_id":"2003.06050","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-curriculum-learning-for-deep-rl-a","title":"Automatic Curriculum Learning For Deep RL: A Short Survey","date":"2020-03-10","arxiv_id":"2003.04664","repositories_listed":0,"syntology":null},{"url":null,"slug":"curriculum-learning-for-reinforcement","title":"Curriculum Learning for Reinforcement Learning Domains: A Framework and Survey","date":"2020-03-10","arxiv_id":"2003.04960","repositories_listed":0,"syntology":null},{"url":null,"slug":"privacy-cost-management-in-smart-meters-using","title":"Privacy-Cost Management in Smart Meters Using Deep Reinforcement Learning","date":"2020-03-10","arxiv_id":"2003.04946","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-mitigating","title":"Reinforcement Learning for Mitigating Intermittent Interference in Terahertz Communication Networks","date":"2020-03-10","arxiv_id":"2003.04832","repositories_listed":0,"syntology":null},{"url":null,"slug":"squirl-robust-and-efficient-learning-from","title":"SQUIRL: Robust and Efficient Learning from Video Demonstration of Long-Horizon Robotic Manipulation Tasks","date":"2020-03-10","arxiv_id":"2003.04956","repositories_listed":0,"syntology":null},{"url":"/paper/the-minerl-competition-on-sample-efficient-1","slug":"the-minerl-competition-on-sample-efficient-1","title":"Retrospective Analysis of the 2019 MineRL Competition on Sample Efficient Reinforcement Learning","date":"2020-03-10","arxiv_id":"2003.05012","repositories_listed":0,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-minerl-competition-on-sample-efficient-1#ran","syntology_url":"https://syntology.ai/paper/2003.05012","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.05012"}},"official":null}},{"url":null,"slug":"advancing-renewable-electricity-consumption","title":"Advancing Renewable Electricity Consumption With Reinforcement Learning","date":"2020-03-09","arxiv_id":"2003.04310","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-ai-interaction-loop-training-new","title":"Human AI interaction loop training: New approach for interactive reinforcement learning","date":"2020-03-09","arxiv_id":"2003.04203","repositories_listed":0,"syntology":null},{"url":null,"slug":"qstar-approximation-schemes-for-batch","title":"Q* Approximation Schemes for Batch Reinforcement Learning: A Theoretical Comparison","date":"2020-03-09","arxiv_id":"2003.03924","repositories_listed":0,"syntology":null},{"url":null,"slug":"transfer-reinforcement-learning-under","title":"Transfer Reinforcement Learning under Unobserved Contextual Information","date":"2020-03-09","arxiv_id":"2003.04427","repositories_listed":0,"syntology":null},{"url":null,"slug":"zooming-for-efficient-model-free","title":"Zooming for Efficient Model-Free Reinforcement Learning in Metric Spaces","date":"2020-03-09","arxiv_id":"2003.04069","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-adversarial-reinforcement-learning-for","title":"Deep Adversarial Reinforcement Learning for Object Disentangling","date":"2020-03-08","arxiv_id":"2003.03779","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-adversarial-imitation-learning-2","title":"Generative Adversarial Imitation Learning with Neural Networks: Global Optimality and Convergence Rate","date":"2020-03-08","arxiv_id":"2003.03709","repositories_listed":0,"syntology":null},{"url":null,"slug":"convergence-of-q-value-in-case-of-gaussian","title":"Convergence of Q-value in case of Gaussian rewards","date":"2020-03-07","arxiv_id":"2003.03526","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-combinatorial","title":"Reinforcement Learning for Combinatorial Optimization: A Survey","date":"2020-03-07","arxiv_id":"2003.03600","repositories_listed":0,"syntology":null},{"url":null,"slug":"cost-sensitive-portfolio-selection-via-deep","title":"Cost-Sensitive Portfolio Selection via Deep Reinforcement Learning","date":"2020-03-06","arxiv_id":"2003.03051","repositories_listed":0,"syntology":null},{"url":null,"slug":"lane-merging-using-policy-based-reinforcement","title":"Lane-Merging Using Policy-based Reinforcement Learning and Post-Optimization","date":"2020-03-06","arxiv_id":"2003.03168","repositories_listed":0,"syntology":null},{"url":null,"slug":"smart-train-operation-algorithms-based-on","title":"Smart Train Operation Algorithms based on Expert Knowledge and Reinforcement Learning","date":"2020-03-06","arxiv_id":"2003.03327","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-robust","title":"Deep Reinforcement Learning-BasedRobust Protection in DER-Rich Distribution Grids","date":"2020-03-05","arxiv_id":"2003.02422","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributional-robustness-and-regularization","title":"Distributional Robustness and Regularization in Reinforcement Learning","date":"2020-03-05","arxiv_id":"2003.02894","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-and-effective-similar-subtrajectory","title":"Efficient and Effective Similar Subtrajectory Search with Deep Reinforcement Learning","date":"2020-03-05","arxiv_id":"2003.02542","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-design-in-cooperative-multi-agent-1","title":"Reward Design in Cooperative Multi-agent Reinforcement Learning for Packet Routing","date":"2020-03-05","arxiv_id":"2003.03433","repositories_listed":0,"syntology":null},{"url":null,"slug":"privacy-aware-time-series-data-sharing-with","title":"Privacy-Aware Time-Series Data Sharing with Deep Reinforcement Learning","date":"2020-03-04","arxiv_id":"2003.02685","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-qos","title":"Deep Reinforcement Learning for QoS-Constrained Resource Allocation in Multiservice Networks","date":"2020-03-03","arxiv_id":"2003.02643","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-exploration-in-constrained","title":"Efficient Exploration in Constrained Environments with Goal-Oriented Reference Path","date":"2020-03-03","arxiv_id":"2003.01641","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-context-aware-task-reasoning-for","title":"Learning Context-aware Task Reasoning for Efficient Meta-reinforcement Learning","date":"2020-03-03","arxiv_id":"2003.01373","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-for-autonomous","title":"Safe Reinforcement Learning for Autonomous Vehicles through Parallel Constrained Policy Optimization","date":"2020-03-03","arxiv_id":"2003.01303","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-object-level-deep","title":"Relevance-Guided Modeling of Object Dynamics for Reinforcement Learning","date":"2020-03-03","arxiv_id":"2003.01384","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-structural-hyper-parameter","title":"Adaptive Structural Hyper-Parameter Configuration by Q-Learning","date":"2020-03-02","arxiv_id":"2003.00863","repositories_listed":0,"syntology":null},{"url":null,"slug":"cluster-based-social-reinforcement-learning","title":"Cluster-Based Social Reinforcement Learning","date":"2020-03-02","arxiv_id":"2003.00627","repositories_listed":0,"syntology":null},{"url":null,"slug":"formal-controller-synthesis-for-continuous","title":"Formal Controller Synthesis for Continuous-Space MDPs via Model-Free Reinforcement Learning","date":"2020-03-02","arxiv_id":"2003.00712","repositories_listed":0,"syntology":null},{"url":null,"slug":"gaussian-process-policy-optimization","title":"Gaussian Process Policy Optimization","date":"2020-03-02","arxiv_id":"2003.01074","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-world-human-robot-collaborative","title":"Real-World Human-Robot Collaborative Reinforcement Learning","date":"2020-03-02","arxiv_id":"2003.01156","repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-averse-learning-by-temporal-difference","title":"Risk-Averse Learning by Temporal Difference Methods","date":"2020-03-02","arxiv_id":"2003.00780","repositories_listed":0,"syntology":null},{"url":null,"slug":"upper-confidence-primal-dual-optimization","title":"Upper Confidence Primal-Dual Reinforcement Learning for CMDP with Adversarial Loss","date":"2020-03-02","arxiv_id":"2003.00660","repositories_listed":0,"syntology":null},{"url":null,"slug":"v2i-connectivity-based-dynamic-queue-jumper","title":"Dynamic Queue-Jump Lane for Emergency Vehicles under Partially Connected Settings: A Multi-Agent Deep Reinforcement Learning Approach","date":"2020-03-02","arxiv_id":"2003.01025","repositories_listed":0,"syntology":null},{"url":null,"slug":"asynchronous-policy-evaluation-in-distributed","title":"Fully Asynchronous Policy Evaluation in Distributed Reinforcement Learning over Networks","date":"2020-03-01","arxiv_id":"2003.00433","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-policy-reuse-using-deep-mixture","title":"Contextual Policy Transfer in Reinforcement Learning Domains via Deep Mixtures-of-Experts","date":"2020-02-29","arxiv_id":"2003.00203","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-near-optimal-policies-with-low","title":"Learning Near Optimal Policies with Low Inherent Bellman Error","date":"2020-02-29","arxiv_id":"2003.00153","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixed-reinforcement-learning-with-additive","title":"Mixed Reinforcement Learning with Additive Stochastic Uncertainty","date":"2020-02-28","arxiv_id":"2003.00848","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-flipit","title":"Deep Reinforcement Learning for FlipIt Security Game","date":"2020-02-28","arxiv_id":"2002.12909","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-through-active","title":"Reinforcement Learning through Active Inference","date":"2020-02-28","arxiv_id":"2002.12636","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-tuning-deep-reinforcement-learning","title":"A Self-Tuning Actor-Critic Algorithm","date":"2020-02-28","arxiv_id":"2002.12928","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-visual-communication-map-for-multi-agent","title":"A Visual Communication Map for Multi-Agent Deep Reinforcement Learning","date":"2020-02-27","arxiv_id":"2002.11882","repositories_listed":0,"syntology":null},{"url":null,"slug":"assembly-robots-with-optimized-control","title":"Assembly robots with optimized control stiffness through reinforcement learning","date":"2020-02-27","arxiv_id":"2002.12207","repositories_listed":0,"syntology":null},{"url":null,"slug":"cautious-reinforcement-learning-via","title":"Cautious Reinforcement Learning via Distributional Risk in the Dual Domain","date":"2020-02-27","arxiv_id":"2002.12475","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-intelligent","title":"Deep Reinforcement Learning Based Intelligent Reflecting Surface for Secure Wireless Communications","date":"2020-02-27","arxiv_id":"2002.12271","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-in-markov-decision-processes-under","title":"Learning in Markov Decision Processes under Constraints","date":"2020-02-27","arxiv_id":"2002.12435","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-resolve-alliance-dilemmas-in-many","title":"Learning to Resolve Alliance Dilemmas in Many-Player Zero-Sum Games","date":"2020-02-27","arxiv_id":"2003.00799","repositories_listed":0,"syntology":null},{"url":null,"slug":"review-analyze-and-design-a-comprehensive","title":"Review, Analysis and Design of a Comprehensive Deep Reinforcement Learning Framework","date":"2020-02-27","arxiv_id":"2002.11883","repositories_listed":0,"syntology":null},{"url":null,"slug":"sub-goal-trees-a-framework-for-goal-based","title":"Sub-Goal Trees -- a Framework for Goal-Based Reinforcement Learning","date":"2020-02-27","arxiv_id":"2002.12361","repositories_listed":0,"syntology":null},{"url":null,"slug":"cautious-reinforcement-learning-with-logical","title":"Cautious Reinforcement Learning with Logical Constraints","date":"2020-02-26","arxiv_id":"2002.12156","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalized-hindsight-for-reinforcement","title":"Generalized Hindsight for Reinforcement Learning","date":"2020-02-26","arxiv_id":"2002.11708","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-ordinary-differential-equation-value","title":"Neural Ordinary Differential Equation Value Networks for Parametrized Action Spaces","date":"2020-02-26","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"when-do-drivers-concentrate-attention-based","title":"When Do Drivers Concentrate? Attention-based Driver Behavior Modeling With Deep Reinforcement Learning","date":"2020-02-26","arxiv_id":"2002.11385","repositories_listed":0,"syntology":null},{"url":null,"slug":"g-learner-and-girl-goal-based-wealth","title":"G-Learner and GIRL: Goal Based Wealth Management with Reinforcement Learning","date":"2020-02-25","arxiv_id":"2002.10990","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-reinforcement-learning-for-2","title":"Model-Based Reinforcement Learning for Physical Systems Without Velocity and Acceleration Measurements","date":"2020-02-25","arxiv_id":"2002.10621","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-reinforcement-learning-for-turn-based-zero","title":"On Reinforcement Learning for Turn-based Zero-sum Markov Games","date":"2020-02-25","arxiv_id":"2002.10620","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-multi-task-imitation-learning-with","title":"Scalable Multi-Task Imitation Learning with Autonomous Improvement","date":"2020-02-25","arxiv_id":"2003.02636","repositories_listed":0,"syntology":null},{"url":null,"slug":"simultaneously-evolving-deep-reinforcement","title":"Simultaneously Evolving Deep Reinforcement Learning Models using Multifactorial Optimization","date":"2020-02-25","arxiv_id":"2002.12133","repositories_listed":0,"syntology":null},{"url":null,"slug":"backpropamine-training-self-modifying-neural-1","title":"Backpropamine: training self-modifying neural networks with differentiable neuromodulated plasticity","date":"2020-02-24","arxiv_id":"2002.10585","repositories_listed":0,"syntology":null},{"url":null,"slug":"millimeter-wave-communications-with-an","title":"Millimeter Wave Communications with an Intelligent Reflector: Performance Optimization and Distributional Reinforcement Learning","date":"2020-02-24","arxiv_id":"2002.10572","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-multi-agent-inverse-reinforcement","title":"Scalable Multi-Agent Inverse Reinforcement Learning via Actor-Attention-Critic","date":"2020-02-24","arxiv_id":"2002.10525","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-linear","title":"Deep Reinforcement Learning with Linear Quadratic Regulator Regions","date":"2020-02-23","arxiv_id":"2002.09820","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-traffic-lights-with-multi-agent","title":"Optimizing Traffic Lights with Multi-agent Deep Reinforcement Learning and V2X communication","date":"2020-02-23","arxiv_id":"2002.09853","repositories_listed":0,"syntology":null},{"url":null,"slug":"wireless-20-towards-an-intelligent-radio","title":"Wireless 2.0: Towards an Intelligent Radio Environment Empowered by Reconfigurable Meta-Surfaces and Artificial Intelligence","date":"2020-02-23","arxiv_id":"2002.11040","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-data-augmentation-via-deep","title":"Automatic Data Augmentation via Deep Reinforcement Learning for Effective Kidney Tumor Segmentation","date":"2020-02-22","arxiv_id":"2002.09703","repositories_listed":0,"syntology":null},{"url":null,"slug":"vehicle-tracking-in-wireless-sensor-networks","title":"Vehicle Tracking in Wireless Sensor Networks via Deep Reinforcement Learning","date":"2020-02-22","arxiv_id":"2002.09671","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-reinforcement-learning-with-a","title":"Accelerating Reinforcement Learning with a Directional-Gaussian-Smoothing Evolution Strategy","date":"2020-02-21","arxiv_id":"2002.09077","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-freshness-and-energy-efficient-uav","title":"Data Freshness and Energy-Efficient UAV Navigation Optimization: A Deep Reinforcement Learning Approach","date":"2020-02-21","arxiv_id":"2003.04816","repositories_listed":0,"syntology":null},{"url":null,"slug":"disentangling-controllable-object-through","title":"Disentangling Controllable Object through Video Prediction Improves Visual Reinforcement Learning","date":"2020-02-21","arxiv_id":"2002.09136","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-search-for-feedback-in-reinforcement","title":"On the Search for Feedback in Reinforcement Learning","date":"2020-02-21","arxiv_id":"2002.09478","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-temporal-difference-learning-with","title":"Adaptive Temporal Difference Learning with Linear Function Approximation","date":"2020-02-20","arxiv_id":"2002.08537","repositories_listed":0,"syntology":null},{"url":"/paper/automatic-gesture-recognition-in-robot","slug":"automatic-gesture-recognition-in-robot","title":"Automatic Gesture Recognition in Robot-assisted Surgery with Reinforcement Learning and Tree Search","date":"2020-02-20","arxiv_id":"2002.08718","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhanced-adversarial-strategically-timed","title":"Enhanced Adversarial Strategically-Timed Attacks against Deep Reinforcement Learning","date":"2020-02-20","arxiv_id":"2002.09027","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-meta-reinforcement-learning-for","title":"Multi-Agent Meta-Reinforcement Learning for Self-Powered and Sustainable Edge Computing Systems","date":"2020-02-20","arxiv_id":"2002.08567","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-as-a","title":"Multi-Agent Reinforcement Learning as a Computational Tool for Language Evolution Research: Historical Context and Future Challenges","date":"2020-02-20","arxiv_id":"2002.08878","repositories_listed":0,"syntology":null},{"url":null,"slug":"oirl-robust-adversarial-inverse-reinforcement","title":"oIRL: Robust Adversarial Inverse Reinforcement Learning with Temporally Extended Actions","date":"2020-02-20","arxiv_id":"2002.09043","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-counterfactual-reinforcement-learning","title":"Debiased Off-Policy Evaluation for Recommendation Systems","date":"2020-02-20","arxiv_id":"2002.08536","repositories_listed":0,"syntology":null},{"url":null,"slug":"curriculum-in-gradient-based-meta","title":"Curriculum in Gradient-Based Meta-Reinforcement Learning","date":"2020-02-19","arxiv_id":"2002.07956","repositories_listed":0,"syntology":null},{"url":null,"slug":"keep-doing-what-worked-behavioral-modelling","title":"Keep Doing What Worked: Behavioral Modelling Priors for Offline Reinforcement Learning","date":"2020-02-19","arxiv_id":"2002.08396","repositories_listed":0,"syntology":null},{"url":null,"slug":"uav-aided-search-and-rescue-operation-using","title":"UAV Aided Search and Rescue Operation Using Reinforcement Learning","date":"2020-02-19","arxiv_id":"2002.08415","repositories_listed":0,"syntology":null},{"url":null,"slug":"empirical-policy-evaluation-with-supergraphs","title":"Empirical Policy Evaluation with Supergraphs","date":"2020-02-18","arxiv_id":"2002.07905","repositories_listed":0,"syntology":null},{"url":null,"slug":"kogun-accelerating-deep-reinforcement","title":"KoGuN: Accelerating Deep Reinforcement Learning via Integrating Human Suboptimal Knowledge","date":"2020-02-18","arxiv_id":"2002.07418","repositories_listed":0,"syntology":null}],"record_sha256":"e3fcb4f685ab535830cbf6394a40dca87d83dc1df1fdd696f49c93238bef881f","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}