{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/106","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":106,"pages_in_order":135,"rows_per_page":100,"rows":[10501,10600],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/105","next":"/task/reinforcement-learning-2/papers/107","papers":[{"url":null,"slug":"a-survey-on-reinforcement-learning-for","title":"A Survey on Reinforcement Learning for Combinatorial Optimization","date":"2020-08-17","arxiv_id":"2008.12248","repositories_listed":0,"syntology":null},{"url":null,"slug":"deepslicing-deep-reinforcement-learning","title":"DeepSlicing: Deep Reinforcement Learning Assisted Resource Allocation for Network Slicing","date":"2020-08-17","arxiv_id":"2008.07614","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-design-by-reinforcement-learning","title":"Generative Design by Reinforcement Learning: Enhancing the Diversity of Topology Optimization Designs","date":"2020-08-17","arxiv_id":"2008.07119","repositories_listed":0,"syntology":null},{"url":null,"slug":"imitation-learning-based-on-entropy","title":"Forward and inverse reinforcement learning sharing network weights and hyperparameters","date":"2020-08-17","arxiv_id":"2008.07284","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-reference-reinforcement-learning-for","title":"Model-Reference Reinforcement Learning for Collision-Free Tracking Control of Autonomous Surface Vehicles","date":"2020-08-17","arxiv_id":"2008.07240","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-sample-complexity-of-reinforcement","title":"On the Sample Complexity of Reinforcement Learning with Policy Space Generalization","date":"2020-08-17","arxiv_id":"2008.07353","repositories_listed":0,"syntology":null},{"url":null,"slug":"playing-catan-with-cross-dimensional-neural","title":"Playing Catan with Cross-dimensional Neural Network","date":"2020-08-17","arxiv_id":"2008.07079","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-adaptive-synchronization-approach-for","title":"An adaptive synchronization approach for weights of deep reinforcement learning","date":"2020-08-16","arxiv_id":"2008.06973","repositories_listed":0,"syntology":null},{"url":null,"slug":"drl-based-qos-aware-resource-allocation","title":"DRL-Based QoS-Aware Resource Allocation Scheme for Coexistence of Licensed and Unlicensed Users in LTE and Beyond","date":"2020-08-16","arxiv_id":"2008.06905","repositories_listed":0,"syntology":null},{"url":null,"slug":"inverse-reinforcement-learning-with-natural","title":"Inverse Reinforcement Learning with Natural Language Goals","date":"2020-08-16","arxiv_id":"2008.06924","repositories_listed":0,"syntology":null},{"url":null,"slug":"chrome-dino-run-using-reinforcement-learning","title":"Chrome Dino Run using Reinforcement Learning","date":"2020-08-15","arxiv_id":"2008.06799","repositories_listed":0,"syntology":null},{"url":null,"slug":"explainability-in-deep-reinforcement-learning","title":"Explainability in Deep Reinforcement Learning","date":"2020-08-15","arxiv_id":"2008.06693","repositories_listed":0,"syntology":null},{"url":null,"slug":"decision-making-at-unsignalized-intersection","title":"Decision-making at Unsignalized Intersection for Autonomous Vehicles: Left-turn Maneuver with Deep Reinforcement Learning","date":"2020-08-14","arxiv_id":"2008.06595","repositories_listed":0,"syntology":null},{"url":null,"slug":"defending-adversarial-attacks-without","title":"Adversary Agnostic Robust Deep Reinforcement Learning","date":"2020-08-14","arxiv_id":"2008.06199","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-trajectory","title":"Reinforcement Learning with Trajectory Feedback","date":"2020-08-13","arxiv_id":"2008.06036","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-ocular-biomechanics-environment-for","title":"An ocular biomechanics environment for reinforcement learning","date":"2020-08-12","arxiv_id":"2008.05088","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-smart-1","title":"A Review of Deep Reinforcement Learning for Smart Building Energy Management","date":"2020-08-12","arxiv_id":"2008.05074","repositories_listed":0,"syntology":null},{"url":null,"slug":"overcoming-model-bias-for-robust-offline-deep","title":"Overcoming Model Bias for Robust Offline Deep Reinforcement Learning","date":"2020-08-12","arxiv_id":"2008.05533","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-intelligent-control-strategy-for-buck-dc","title":"An Intelligent Control Strategy for buck DC-DC Converter via Deep Reinforcement Learning","date":"2020-08-11","arxiv_id":"2008.04542","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-deep-reinforcement-learning-for-1","title":"Deep Model-Based Reinforcement Learning for High-Dimensional Problems, a Survey","date":"2020-08-11","arxiv_id":"2008.05598","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparison-of-model-predictive-and","title":"Comparison of Model Predictive and Reinforcement Learning Methods for Fault Tolerant Control","date":"2020-08-10","arxiv_id":"2008.04403","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-label","title":"Deep Reinforcement Learning with Label Embedding Reward for Supervised Image Hashing","date":"2020-08-10","arxiv_id":"2008.03973","repositories_listed":0,"syntology":null},{"url":null,"slug":"fault-tolerant-control-of-degrading-systems","title":"Fault-Tolerant Control of Degrading Systems with On-Policy Reinforcement Learning","date":"2020-08-10","arxiv_id":"2008.04407","repositories_listed":0,"syntology":null},{"url":null,"slug":"grimgep-learning-progress-for-robust-goal","title":"GRIMGEP: Learning Progress for Robust Goal Sampling in Visual Deep Reinforcement Learning","date":"2020-08-10","arxiv_id":"2008.04388","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchial-reinforcement-learning-in","title":"Hierarchical Reinforcement Learning in StarCraft II with Human Expertise in Subgoals Selection","date":"2020-08-08","arxiv_id":"2008.03444","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-machine-of-few-words-interactive-speaker","title":"A Machine of Few Words -- Interactive Speaker Recognition with Reinforcement Learning","date":"2020-08-07","arxiv_id":"2008.03127","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-deep-reinforcement-learning-for-1","title":"Distributed Deep Reinforcement Learning for Functional Split Control in Energy Harvesting Virtualized Small Cells","date":"2020-08-07","arxiv_id":"2008.04105","repositories_listed":0,"syntology":null},{"url":null,"slug":"incremental-text-to-speech-for-neural","title":"Incremental Text to Speech for Neural Sequence-to-Sequence Models using Reinforcement Learning","date":"2020-08-07","arxiv_id":"2008.03096","repositories_listed":0,"syntology":null},{"url":null,"slug":"managing-caching-strategies-for-stream","title":"Managing caching strategies for stream reasoning with reinforcement learning","date":"2020-08-07","arxiv_id":"2008.03212","repositories_listed":0,"syntology":null},{"url":null,"slug":"physics-based-dexterous-manipulations-with","title":"Physics-Based Dexterous Manipulations with Estimated Hand Poses and Residual Reinforcement Learning","date":"2020-08-07","arxiv_id":"2008.03285","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-gentle-lecture-note-on-filtrations-in","title":"A Gentle Lecture Note on Filtrations in Reinforcement Learning","date":"2020-08-06","arxiv_id":"2008.02622","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-coordination-offsets-for-signalized","title":"Adaptive Coordination Offsets for Signalized Arterial Intersections using Deep Reinforcement Learning","date":"2020-08-06","arxiv_id":"2008.02691","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-q-network-based-multi-agent","title":"Deep Q-Network Based Multi-agent Reinforcement Learning with Binary Action Agents","date":"2020-08-06","arxiv_id":"2008.04109","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-to-detect-brain","title":"Deep reinforcement learning to detect brain lesions on MRI: a proof-of-concept application of reinforcement learning to medical images","date":"2020-08-06","arxiv_id":"2008.02708","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-field","title":"Deep Reinforcement Learning for Field Development Optimization","date":"2020-08-05","arxiv_id":"2008.12627","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-power-control-from-a-fixed-batch-of","title":"Learning Power Control from a Fixed Batch of Data","date":"2020-08-05","arxiv_id":"2008.02669","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-ad-pruning-of-sponsored-search","title":"Optimizing AD Pruning of Sponsored Search with Reinforcement Learning","date":"2020-08-05","arxiv_id":"2008.02014","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-driven-information","title":"Reinforcement Learning-driven Information Seeking: A Quantum Probabilistic Approach","date":"2020-08-05","arxiv_id":"2008.02372","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparative-analysis-of-deep-reinforcement","title":"A Comparative Analysis of Deep Reinforcement Learning-enabled Freeway Decision-making for Automated Vehicles","date":"2020-08-04","arxiv_id":"2008.01302","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-relearning-approach-to-reinforcement","title":"A Relearning Approach to Reinforcement Learning for Control of Smart Buildings","date":"2020-08-04","arxiv_id":"2008.01879","repositories_listed":0,"syntology":null},{"url":null,"slug":"easyrl-a-simple-and-extensible-reinforcement","title":"EasyRL: A Simple and Extensible Reinforcement Learning Framework","date":"2020-08-04","arxiv_id":"2008.01700","repositories_listed":0,"syntology":null},{"url":null,"slug":"explanation-of-reinforcement-learning-model","title":"Explanation of Reinforcement Learning Model in Dynamic Multi-Agent System","date":"2020-08-04","arxiv_id":"2008.01508","repositories_listed":0,"syntology":null},{"url":null,"slug":"faded-experience-trust-region-policy","title":"Faded-Experience Trust Region Policy Optimization for Model-Free Power Allocation in Interference Channel","date":"2020-08-04","arxiv_id":"2008.01705","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-transition-models-with-time-delayed","title":"Learning Transition Models with Time-delayed Causal Relations","date":"2020-08-04","arxiv_id":"2008.01593","repositories_listed":0,"syntology":null},{"url":null,"slug":"cooperative-control-of-mobile-robots-with","title":"Cooperative Control of Mobile Robots with Stackelberg Learning","date":"2020-08-03","arxiv_id":"2008.00679","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamics-generalization-via-information","title":"Dynamics Generalization via Information Bottleneck in Deep Reinforcement Learning","date":"2020-08-03","arxiv_id":"2008.00614","repositories_listed":0,"syntology":null},{"url":null,"slug":"fully-decentralized-reinforcement-learning","title":"Fully Decentralized Reinforcement Learning-based Control of Photovoltaics in Distribution Grids for Joint Provision of Real and Reactive Power","date":"2020-08-03","arxiv_id":"2008.01231","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-play-two-player-perfect","title":"Learning to Play Two-Player Perfect-Information Games without Knowledge","date":"2020-08-03","arxiv_id":"2008.01188","repositories_listed":0,"syntology":null},{"url":null,"slug":"proximal-deterministic-policy-gradient","title":"Proximal Deterministic Policy Gradient","date":"2020-08-03","arxiv_id":"2008.00759","repositories_listed":0,"syntology":null},{"url":null,"slug":"tracking-the-race-between-deep-reinforcement","title":"Tracking the Race Between Deep Reinforcement Learning and Imitation Learning -- Extended Version","date":"2020-08-03","arxiv_id":"2008.00766","repositories_listed":0,"syntology":null},{"url":null,"slug":"curriculum-learning-with-a-progression","title":"Curriculum Learning with a Progression Function","date":"2020-08-02","arxiv_id":"2008.00511","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-mobile-edge","title":"Deep Reinforcement Learning Based Mobile Edge Computing for Intelligent Internet of Things","date":"2020-08-01","arxiv_id":"2008.00250","repositories_listed":0,"syntology":null},{"url":null,"slug":"ergodic-annealing","title":"Ergodic Annealing","date":"2020-08-01","arxiv_id":"2008.00234","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-with-safety-constraints-sample","title":"Learning with Safety Constraints: Sample Complexity of Reinforcement Learning for Constrained MDPs","date":"2020-08-01","arxiv_id":"2008.00311","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-batch-sampling-with-reinforcement","title":"Neural Batch Sampling with Reinforcement Learning for Semi-Supervised Anomaly Detection","date":"2020-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"spatial-geometric-reasoning-for-room-layout","title":"Spatial Geometric Reasoning for Room Layout Estimation via Deep Reinforcement Learning","date":"2020-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-using-cyclical","title":"Deep Reinforcement Learning using Cyclical Learning Rates","date":"2020-07-31","arxiv_id":"2008.01171","repositories_listed":0,"syntology":null},{"url":null,"slug":"intelligentpooling-practical-thompson","title":"IntelligentPooling: Practical Thompson Sampling for mHealth","date":"2020-07-31","arxiv_id":"2008.01571","repositories_listed":0,"syntology":null},{"url":null,"slug":"chance-constrained-policy-optimization-for","title":"Chance Constrained Policy Optimization for Process Control and Optimization","date":"2020-07-30","arxiv_id":"2008.00030","repositories_listed":0,"syntology":null},{"url":null,"slug":"mapper-multi-agent-path-planning-with","title":"MAPPER: Multi-Agent Path Planning with Evolutionary Reinforcement Learning in Mixed Dynamic Environments","date":"2020-07-30","arxiv_id":"2007.15724","repositories_listed":0,"syntology":null},{"url":null,"slug":"moody-learners-explaining-competitive","title":"Moody Learners -- Explaining Competitive Behaviour of Reinforcement Learning Agents","date":"2020-07-30","arxiv_id":"2007.16045","repositories_listed":0,"syntology":null},{"url":null,"slug":"compare-and-select-video-summarization-with","title":"Compare and Select: Video Summarization with Multi-Agent Reinforcement Learning","date":"2020-07-29","arxiv_id":"2007.14552","repositories_listed":0,"syntology":null},{"url":null,"slug":"dreaming-model-based-reinforcement-learning","title":"Dreaming: Model-based Reinforcement Learning by Latent Imagination without Reconstruction","date":"2020-07-29","arxiv_id":"2007.14535","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantity-vs-quality-on-hyperparameter","title":"Quantity vs. Quality: On Hyperparameter Optimization for Deep Reinforcement Learning","date":"2020-07-29","arxiv_id":"2007.14604","repositories_listed":0,"syntology":null},{"url":null,"slug":"cooperative-internet-of-uavs-distributed","title":"Cooperative Internet of UAVs: Distributed Trajectory Design by Multi-agent Deep Reinforcement Learning","date":"2020-07-28","arxiv_id":"2007.14297","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-control-of-multi-agent-systems","title":"Hierarchical Control of Multi-Agent Systems using Online Reinforcement Learning","date":"2020-07-28","arxiv_id":"2007.14186","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-height-optimisation-for-cellular","title":"Adaptive Height Optimisation for Cellular-Connected UAVs using Reinforcement Learning","date":"2020-07-27","arxiv_id":"2007.13695","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-active-learning-for-pure-exploration-in","title":"Fast active learning for pure exploration in reinforcement learning","date":"2020-07-27","arxiv_id":"2007.13442","repositories_listed":0,"syntology":null},{"url":null,"slug":"greedy-bandits-with-sampled-context","title":"Greedy Bandits with Sampled Context","date":"2020-07-27","arxiv_id":"2007.16001","repositories_listed":0,"syntology":null},{"url":null,"slug":"intelligent-trajectory-planning-in-uav","title":"Intelligent Trajectory Planning in UAV-mounted Wireless Networks: A Quantum-Inspired Reinforcement Learning Perspective","date":"2020-07-27","arxiv_id":"2007.13418","repositories_listed":0,"syntology":null},{"url":null,"slug":"off-policy-evaluation-in-infinite-horizon","title":"Off-policy Evaluation in Infinite-Horizon Reinforcement Learning with Latent Confounders","date":"2020-07-27","arxiv_id":"2007.13893","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-efficient-visuomotor-policy-training","title":"Data-efficient visuomotor policy training using reinforcement learning and generative models","date":"2020-07-26","arxiv_id":"2007.13134","repositories_listed":0,"syntology":null},{"url":null,"slug":"automated-database-indexing-using-model-free","title":"Automated Database Indexing using Model-free Reinforcement Learning","date":"2020-07-25","arxiv_id":"2007.14244","repositories_listed":0,"syntology":null},{"url":null,"slug":"variance-reduction-for-deep-q-learning-using","title":"Variance Reduction for Deep Q-Learning using Stochastic Recursive Gradient","date":"2020-07-25","arxiv_id":"2007.12817","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparative-study-of-ai-based-intrusion","title":"A Comparative Study of AI-based Intrusion Detection Techniques in Critical Infrastructures","date":"2020-07-24","arxiv_id":"2008.00088","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-safety-aware-model-based-reinforcement","title":"Safe Model-Based Reinforcement Learning for Systems with Parametric Uncertainties","date":"2020-07-24","arxiv_id":"2007.12666","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-energy-management-for-real-driving","title":"Adaptive Energy Management for Real Driving Conditions via Transfer Reinforcement Learning","date":"2020-07-24","arxiv_id":"2007.12560","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrated-longitudinal-speed-decision-making","title":"Integrated Longitudinal Speed Decision-Making and Energy Efficiency Control for Connected Electrified Vehicles","date":"2020-07-24","arxiv_id":"2007.12565","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-the-imitation-gap-by-adaptive","title":"Bridging the Imitation Gap by Adaptive Insubordination","date":"2020-07-23","arxiv_id":"2007.12173","repositories_listed":0,"syntology":null},{"url":null,"slug":"explore-more-and-improve-regret-in-linear","title":"Reinforcement Learning with Fast Stabilization in Linear Dynamical Systems","date":"2020-07-23","arxiv_id":"2007.12291","repositories_listed":0,"syntology":null},{"url":null,"slug":"buffer-pool-aware-query-scheduling-via-deep","title":"Buffer Pool Aware Query Scheduling via Deep Reinforcement Learning","date":"2020-07-21","arxiv_id":"2007.10568","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-convergence-of-reinforcement-learning","title":"On the Convergence of Reinforcement Learning with Monte Carlo Exploring Starts","date":"2020-07-21","arxiv_id":"2007.10916","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-short-note-on-soft-max-and-policy-gradients","title":"A Short Note on Soft-max and Policy Gradients in Bandits Problems","date":"2020-07-20","arxiv_id":"2007.10297","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-control-by-reinforcement","title":"Interpretable Control by Reinforcement Learning","date":"2020-07-20","arxiv_id":"2007.09964","repositories_listed":0,"syntology":null},{"url":null,"slug":"lagrangian-duality-in-reinforcement-learning","title":"Lagrangian Duality in Reinforcement Learning","date":"2020-07-20","arxiv_id":"2007.09998","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-overview-of-natural-language-state","title":"An Overview of Natural Language State Representation for Reinforcement Learning","date":"2020-07-19","arxiv_id":"2007.09774","repositories_listed":0,"syntology":null},{"url":"/paper/catch-context-based-meta-reinforcement","slug":"catch-context-based-meta-reinforcement","title":"CATCH: Context-based Meta Reinforcement Learning for Transferrable Architecture Search","date":"2020-07-18","arxiv_id":"2007.09380","repositories_listed":0,"syntology":null},{"url":null,"slug":"quick-question-interrupting-users-for","title":"Quick Question: Interrupting Users for Microtasks with Reinforcement Learning","date":"2020-07-18","arxiv_id":"2007.09515","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-deep-reinforcement-learning-2","title":"Hierarchical Deep Reinforcement Learning Approach for Multi-Objective Scheduling With Varying Queue Sizes","date":"2020-07-17","arxiv_id":"2007.09256","repositories_listed":0,"syntology":null},{"url":null,"slug":"hyperparameter-selection-for-offline","title":"Hyperparameter Selection for Offline Reinforcement Learning","date":"2020-07-17","arxiv_id":"2007.09055","repositories_listed":0,"syntology":null},{"url":null,"slug":"cones-convex-natural-evolutionary-strategies","title":"CoNES: Convex Natural Evolutionary Strategies","date":"2020-07-16","arxiv_id":"2007.08601","repositories_listed":0,"syntology":null},{"url":null,"slug":"decision-making-strategy-on-highway-for","title":"Decision-making Strategy on Highway for Autonomous Vehicles using Deep Reinforcement Learning","date":"2020-07-16","arxiv_id":"2007.08691","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-reinforcement-learning-of","title":"Distributed Reinforcement Learning of Targeted Grasping with Active Vision for Mobile Manipulators","date":"2020-07-16","arxiv_id":"2007.08082","repositories_listed":0,"syntology":null},{"url":null,"slug":"drift-deep-reinforcement-learning-for","title":"DRIFT: Deep Reinforcement Learning for Functional Software Testing","date":"2020-07-16","arxiv_id":"2007.08220","repositories_listed":0,"syntology":null},{"url":null,"slug":"dueling-deep-q-network-for-highway-decision","title":"Dueling Deep Q Network for Highway Decision Making in Autonomous Vehicles: A Case Study","date":"2020-07-16","arxiv_id":"2007.08343","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-like-energy-management-based-on-deep","title":"Human-like Energy Management Based on Deep Reinforcement Learning and Historical Driving Experiences","date":"2020-07-16","arxiv_id":"2007.10126","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-gradient-reinforcement-learning-with-an","title":"Meta-Gradient Reinforcement Learning with an Objective Discovered Online","date":"2020-07-16","arxiv_id":"2007.08433","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-enabled-decision","title":"Reinforcement Learning-Enabled Decision-Making Strategies for a Vehicle-Cyber-Physical-System in Connected Environment","date":"2020-07-16","arxiv_id":"2007.09101","repositories_listed":0,"syntology":null},{"url":null,"slug":"transfer-deep-reinforcement-learning-enabled","title":"Transfer Deep Reinforcement Learning-enabled Energy Management Strategy for Hybrid Tracked Vehicle","date":"2020-07-16","arxiv_id":"2007.08690","repositories_listed":0,"syntology":null},{"url":null,"slug":"transferred-energy-management-strategies-for","title":"Transferred Energy Management Strategies for Hybrid Electric Vehicles Based on Driving Conditions Recognition","date":"2020-07-16","arxiv_id":"2007.08337","repositories_listed":0,"syntology":null}],"record_sha256":"920e2f9d35b05df26d4a04670a40029477a48269f4b12a8c4d35017648ce6594","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}