{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/117","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":117,"pages_in_order":152,"rows_per_page":100,"rows":[11601,11700],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/116","next":"/task/reinforcement-learning-1/papers/118","papers":[{"url":null,"slug":"guided-policy-search-based-control-of-a-high","title":"Guided Policy Search Based Control of a High Dimensional Advanced Manufacturing Process","date":"2020-09-12","arxiv_id":"2009.05838","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-learning-interference-cancellation-in","title":"Deep Learning Interference Cancellation in Wireless Networks","date":"2020-09-11","arxiv_id":"2009.05533","repositories_listed":0,"syntology":null},{"url":null,"slug":"embodied-visual-navigation-with-automatic","title":"Embodied Visual Navigation with Automatic Curriculum Learning in Real Environments","date":"2020-09-11","arxiv_id":"2009.05429","repositories_listed":0,"syntology":null},{"url":null,"slug":"covid-19-pandemic-cyclic-lockdown","title":"COVID-19 Pandemic Cyclic Lockdown Optimization Using Reinforcement Learning","date":"2020-09-10","arxiv_id":"2009.04647","repositories_listed":0,"syntology":null},{"url":null,"slug":"importance-weighted-policy-learning-and","title":"Importance Weighted Policy Learning and Adaptation","date":"2020-09-10","arxiv_id":"2009.04875","repositories_listed":0,"syntology":null},{"url":null,"slug":"rlcfr-minimize-counterfactual-regret-by-deep","title":"RLCFR: Minimize Counterfactual Regret by Deep Reinforcement Learning","date":"2020-09-10","arxiv_id":"2009.06373","repositories_listed":0,"syntology":null},{"url":null,"slug":"aoi-minimization-in-status-update-control","title":"AoI Minimization in Status Update Control with Energy Harvesting Sensors","date":"2020-09-09","arxiv_id":"2009.04224","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-option","title":"Deep Reinforcement Learning for Option Replication and Hedging","date":"2020-09-09","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-objective-reinforcement-learning-for","title":"Multi-Objective Model-based Reinforcement Learning for Infectious Disease Control","date":"2020-09-09","arxiv_id":"2009.04607","repositories_listed":0,"syntology":null},{"url":null,"slug":"qr-mix-distributional-value-function","title":"QR-MIX: Distributional Value Function Factorisation for Cooperative Multi-Agent Reinforcement Learning","date":"2020-09-09","arxiv_id":"2009.04197","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-non-stationary-1","title":"Reinforcement Learning in Non-Stationary Discrete-Time Linear-Quadratic Mean-Field Games","date":"2020-09-09","arxiv_id":"2009.04350","repositories_listed":0,"syntology":null},{"url":null,"slug":"energy-expenditure-estimation-through-daily","title":"Energy Expenditure Estimation Through Daily Activity Recognition Using a Smart-phone","date":"2020-09-08","arxiv_id":"2009.03681","repositories_listed":0,"syntology":null},{"url":null,"slug":"evolutionary-reinforcement-learning-via","title":"Evolutionary Reinforcement Learning via Cooperative Coevolutionary Negatively Correlated Search","date":"2020-09-08","arxiv_id":"2009.03603","repositories_listed":0,"syntology":null},{"url":null,"slug":"induction-and-exploitation-of-subgoal","title":"Induction and Exploitation of Subgoal Automata for Reinforcement Learning","date":"2020-09-08","arxiv_id":"2009.03855","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-on-job-shop-scheduling","title":"Graph neural networks-based Scheduler for Production planning problems using Reinforcement Learning","date":"2020-09-08","arxiv_id":"2009.03836","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-learning-of-causal-structures-with","title":"Active Learning of Causal Structures with Deep Reinforcement Learning","date":"2020-09-07","arxiv_id":"2009.03009","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-learning-and-reinforcement-learning-for","title":"Deep Learning and Reinforcement Learning for Autonomous Unmanned Aerial Systems: Roadmap for Theory to Deployment","date":"2020-09-07","arxiv_id":"2009.03349","repositories_listed":0,"syntology":null},{"url":null,"slug":"detecting-and-adapting-to-crisis-pattern-with","title":"Detecting and adapting to crisis pattern with context based Deep Reinforcement Learning","date":"2020-09-07","arxiv_id":"2009.07200","repositories_listed":0,"syntology":null},{"url":null,"slug":"driving-tasks-transfer-in-deep-reinforcement","title":"Driving Tasks Transfer in Deep Reinforcement Learning for Decision-making of Autonomous Vehicles","date":"2020-09-07","arxiv_id":"2009.03268","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-spoken-language-understanding-with-rl","title":"Robust Spoken Language Understanding with RL-based Value Error Recovery","date":"2020-09-07","arxiv_id":"2009.03095","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-hybrid-pac-reinforcement-learning-algorithm","title":"A Hybrid PAC Reinforcement Learning Algorithm","date":"2020-09-05","arxiv_id":"2009.02602","repositories_listed":0,"syntology":null},{"url":null,"slug":"pac-reinforcement-learning-algorithm-for","title":"PAC Reinforcement Learning Algorithm for General-Sum Markov Games","date":"2020-09-05","arxiv_id":"2009.02605","repositories_listed":0,"syntology":null},{"url":null,"slug":"visualizing-the-loss-landscape-of-actor","title":"Visualizing the Loss Landscape of Actor Critic Methods with Applications in Inventory Optimization","date":"2020-09-04","arxiv_id":"2009.02391","repositories_listed":0,"syntology":null},{"url":null,"slug":"sparse-meta-networks-for-sequential","title":"Sparse Meta Networks for Sequential Adaptation and its Application to Adaptive Language Modelling","date":"2020-09-03","arxiv_id":"2009.01803","repositories_listed":0,"syntology":null},{"url":null,"slug":"tap-net-transport-and-pack-using","title":"TAP-Net: Transport-and-Pack using Reinforcement Learning","date":"2020-09-03","arxiv_id":"2009.01469","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-approach-to-hybrid","title":"A reinforcement learning approach to hybrid control design","date":"2020-09-02","arxiv_id":"2009.00821","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-reinforcement-learning-model-for","title":"Adaptive Reinforcement Learning Model for Simulation of Urban Mobility during Crises","date":"2020-09-02","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"vulnerability-aware-poisoning-mechanism-for","title":"Vulnerability-Aware Poisoning Mechanism for Online RL with Unknown Dynamics","date":"2020-09-02","arxiv_id":"2009.00774","repositories_listed":0,"syntology":null},{"url":null,"slug":"plotthread-creating-expressive-storyline","title":"PlotThread: Creating Expressive Storyline Visualizations using Reinforcement Learning","date":"2020-09-01","arxiv_id":"2009.00249","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-black-box","title":"Reinforcement Learning-based Black-Box Evasion Attacks to Link Prediction in Dynamic Graphs","date":"2020-09-01","arxiv_id":"2009.00163","repositories_listed":0,"syntology":null},{"url":null,"slug":"solving-the-single-track-train-scheduling","title":"Solving the single-track train scheduling problem via Deep Reinforcement Learning","date":"2020-09-01","arxiv_id":"2009.00433","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-variance-reduction-understanding-the","title":"Beyond variance reduction: Understanding the true impact of baselines on policy optimization","date":"2020-08-31","arxiv_id":"2008.13773","repositories_listed":0,"syntology":null},{"url":null,"slug":"control-of-a-nature-inspired-scorpion-using","title":"Control of a Nature-inspired Scorpion using Reinforcement Learning","date":"2020-08-31","arxiv_id":"2008.13712","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-driven-outer-loop-control-using-deep","title":"Data-driven Outer-Loop Control Using Deep Reinforcement Learning for Trajectory Tracking","date":"2020-08-31","arxiv_id":"2008.13732","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-reinforcement-learning-in-factored","title":"Efficient Reinforcement Learning in Factored MDPs with Application to Constrained RL","date":"2020-08-31","arxiv_id":"2008.13319","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-contact-rich","title":"Deep Reinforcement Learning for Contact-Rich Skills Using Compliant Movement Primitives","date":"2020-08-30","arxiv_id":"2008.13223","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-in-the-loop-methods-for-data-driven-and","title":"Human-in-the-Loop Methods for Data-Driven and Reinforcement Learning Systems","date":"2020-08-30","arxiv_id":"2008.13221","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-does-the-structure-embedded-in-learning","title":"How does the structure embedded in learning policy affect learning quadruped locomotion?","date":"2020-08-29","arxiv_id":"2008.12970","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-feedback","title":"Reinforcement Learning with Feedback-modulated TD-STDP","date":"2020-08-29","arxiv_id":"2008.13044","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-based-lane-change","title":"Meta Reinforcement Learning-Based Lane Change Strategy for Autonomous Vehicles","date":"2020-08-28","arxiv_id":"2008.12451","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-world-video-adaptation-with","title":"Real-world Video Adaptation with Reinforcement Learning","date":"2020-08-28","arxiv_id":"2008.12858","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-efficiency-in-sparse-reinforcement","title":"Sample Efficiency in Sparse Reinforcement Learning: Or Your Money Back","date":"2020-08-28","arxiv_id":"2008.12693","repositories_listed":0,"syntology":null},{"url":null,"slug":"controlling-level-of-unconsciousness-with","title":"Controlling Level of Unconsciousness by Titrating Propofol with Deep Reinforcement Learning","date":"2020-08-27","arxiv_id":"2008.12333","repositories_listed":0,"syntology":null},{"url":null,"slug":"document-editing-assistants-and-model-based","title":"Document-editing Assistants and Model-based Reinforcement Learning as a Path to Conversational AI","date":"2020-08-27","arxiv_id":"2008.12095","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-advantage-regret-matching-actor-critic","title":"The Advantage Regret-Matching Actor-Critic","date":"2020-08-27","arxiv_id":"2008.12234","repositories_listed":0,"syntology":null},{"url":null,"slug":"constrained-markov-decision-processes-via","title":"Constrained Markov Decision Processes via Backward Value Functions","date":"2020-08-26","arxiv_id":"2008.11811","repositories_listed":0,"syntology":null},{"url":null,"slug":"decision-making-for-autonomous-vehicles-on","title":"Decision-making for Autonomous Vehicles on Highway: Deep Reinforcement Learning with Continuous Action Horizon","date":"2020-08-26","arxiv_id":"2008.11852","repositories_listed":0,"syntology":null},{"url":null,"slug":"identifying-critical-states-by-the-action","title":"Identifying Critical States by the Action-Based Variance of Expected Return","date":"2020-08-26","arxiv_id":"2008.11332","repositories_listed":0,"syntology":null},{"url":null,"slug":"selective-particle-attention-visual-feature","title":"Selective Particle Attention: Visual Feature-Based Attention in Deep Reinforcement Learning","date":"2020-08-26","arxiv_id":"2008.11491","repositories_listed":0,"syntology":null},{"url":null,"slug":"synthetic-sample-selection-via-reinforcement","title":"Synthetic Sample Selection via Reinforcement Learning","date":"2020-08-26","arxiv_id":"2008.11331","repositories_listed":0,"syntology":null},{"url":null,"slug":"auxiliary-task-based-deep-reinforcement","title":"Auxiliary-task Based Deep Reinforcement Learning for Participant Selection Problem in Mobile Crowdsourcing","date":"2020-08-25","arxiv_id":"2008.11087","repositories_listed":0,"syntology":null},{"url":null,"slug":"ensuring-monotonic-policy-improvement-in","title":"Ensuring Monotonic Policy Improvement in Entropy-regularized Value-based Reinforcement Learning","date":"2020-08-25","arxiv_id":"2008.10806","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-reinforcement-learning-a-case-study-in","title":"Robust Reinforcement Learning: A Case Study in Linear Quadratic Regulation","date":"2020-08-25","arxiv_id":"2008.11592","repositories_listed":0,"syntology":null},{"url":null,"slug":"t-soft-update-of-target-network-for-deep","title":"t-Soft Update of Target Network for Deep Reinforcement Learning","date":"2020-08-25","arxiv_id":"2008.10861","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-dispatching-for-large-scale","title":"Dynamic Dispatching for Large-Scale Heterogeneous Fleet via Multi-agent Deep Reinforcement Learning","date":"2020-08-24","arxiv_id":"2008.10713","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-memories-learning","title":"Improved Memories Learning","date":"2020-08-24","arxiv_id":"2008.10433","repositories_listed":0,"syntology":null},{"url":null,"slug":"variable-compliance-control-for-robotic-peg","title":"Variable Compliance Control for Robotic Peg-in-Hole Assembly: A Deep Reinforcement Learning Approach","date":"2020-08-24","arxiv_id":"2008.10224","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-and-multiple-time-scale-eligibility","title":"Adaptive and Multiple Time-scale Eligibility Traces for Online Deep Reinforcement Learning","date":"2020-08-23","arxiv_id":"2008.10040","repositories_listed":0,"syntology":null},{"url":null,"slug":"dsp-a-differential-spatial-prediction-scheme","title":"DSP: A Differential Spatial Prediction Scheme for Comprehensive real industrial datasets","date":"2020-08-23","arxiv_id":"2008.09951","repositories_listed":0,"syntology":null},{"url":null,"slug":"mobile-networks-for-computer-go","title":"Mobile Networks for Computer Go","date":"2020-08-23","arxiv_id":"2008.10080","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-imitation-learning-via-random","title":"Adversarial Imitation Learning via Random Search","date":"2020-08-21","arxiv_id":"2008.09450","repositories_listed":0,"syntology":null},{"url":null,"slug":"biomechanic-posture-stabilisation-via","title":"Biomechanic Posture Stabilisation via Iterative Training of Multi-policy Deep Reinforcement Learning Agents","date":"2020-08-21","arxiv_id":"2008.12210","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-collaborate-in-multi-module","title":"Learning to Collaborate in Multi-Module Recommendation via Multi-Agent Reinforcement Learning without Communication","date":"2020-08-21","arxiv_id":"2008.09369","repositories_listed":0,"syntology":null},{"url":"/paper/model-free-episodic-control-with-state","slug":"model-free-episodic-control-with-state","title":"Model-Free Episodic Control with State Aggregation","date":"2020-08-21","arxiv_id":"2008.09685","repositories_listed":0,"syntology":null},{"url":null,"slug":"nancy-neural-adaptive-network-coding","title":"NANCY: Neural Adaptive Network Coding methodologY for video distribution over wireless networks","date":"2020-08-21","arxiv_id":"2008.09559","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-admission","title":"Reinforcement Learning-based Admission Control in Delay-sensitive Service Systems","date":"2020-08-21","arxiv_id":"2008.09590","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-free-optimal-control-of-discrete-time","title":"Model-free optimal control of discrete-time systems with additive and multiplicative noises","date":"2020-08-20","arxiv_id":"2008.08734","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-dynamic-weighing","title":"Reinforcement Learning based dynamic weighing of Ensemble Models for Time Series Forecasting","date":"2020-08-20","arxiv_id":"2008.08878","repositories_listed":0,"syntology":null},{"url":null,"slug":"static-neural-compiler-optimization-via-deep","title":"Static Neural Compiler Optimization via Deep Reinforcement Learning","date":"2020-08-20","arxiv_id":"2008.08951","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-knowledge-based-sequential","title":"A Survey of Knowledge-based Sequential Decision Making under Uncertainty","date":"2020-08-19","arxiv_id":"2008.08548","repositories_listed":0,"syntology":null},{"url":null,"slug":"intelligent-replication-management-for-hdfs","title":"Intelligent Replication Management for HDFS Using Reinforcement Learning","date":"2020-08-19","arxiv_id":"2008.08665","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-framework-for-studying-reinforcement","title":"A Framework for Studying Reinforcement Learning and Sim-to-Real in Robot Soccer","date":"2020-08-18","arxiv_id":"2008.12624","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-trading-strategies-across-liquidity","title":"Adaptive trading strategies across liquidity pools","date":"2020-08-18","arxiv_id":"2008.07807","repositories_listed":0,"syntology":null},{"url":null,"slug":"analysis-of-social-robotic-navigation","title":"Analysis of Social Robotic Navigation approaches: CNN Encoder and Incremental Learning as an alternative to Deep Reinforcement Learning","date":"2020-08-18","arxiv_id":"2008.07965","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-improving-object","title":"Reinforcement Learning for Improving Object Detection","date":"2020-08-18","arxiv_id":"2008.08005","repositories_listed":0,"syntology":null},{"url":null,"slug":"relmogen-leveraging-motion-generation-in","title":"ReLMoGen: Leveraging Motion Generation in Reinforcement Learning for Mobile Manipulation","date":"2020-08-18","arxiv_id":"2008.07792","repositories_listed":0,"syntology":null},{"url":null,"slug":"residual-learning-from-demonstration","title":"Residual Learning from Demonstration: Adapting DMPs for Contact-rich Manipulation","date":"2020-08-18","arxiv_id":"2008.07682","repositories_listed":0,"syntology":null},{"url":null,"slug":"super-human-performance-in-gran-turismo-sport","title":"Super-Human Performance in Gran Turismo Sport Using Deep Reinforcement Learning","date":"2020-08-18","arxiv_id":"2008.07971","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-reinforcement-learning-for","title":"A Survey on Reinforcement Learning for Combinatorial Optimization","date":"2020-08-17","arxiv_id":"2008.12248","repositories_listed":0,"syntology":null},{"url":null,"slug":"deepslicing-deep-reinforcement-learning","title":"DeepSlicing: Deep Reinforcement Learning Assisted Resource Allocation for Network Slicing","date":"2020-08-17","arxiv_id":"2008.07614","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-design-by-reinforcement-learning","title":"Generative Design by Reinforcement Learning: Enhancing the Diversity of Topology Optimization Designs","date":"2020-08-17","arxiv_id":"2008.07119","repositories_listed":0,"syntology":null},{"url":null,"slug":"imitation-learning-based-on-entropy","title":"Forward and inverse reinforcement learning sharing network weights and hyperparameters","date":"2020-08-17","arxiv_id":"2008.07284","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-reference-reinforcement-learning-for","title":"Model-Reference Reinforcement Learning for Collision-Free Tracking Control of Autonomous Surface Vehicles","date":"2020-08-17","arxiv_id":"2008.07240","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-sample-complexity-of-reinforcement","title":"On the Sample Complexity of Reinforcement Learning with Policy Space Generalization","date":"2020-08-17","arxiv_id":"2008.07353","repositories_listed":0,"syntology":null},{"url":null,"slug":"playing-catan-with-cross-dimensional-neural","title":"Playing Catan with Cross-dimensional Neural Network","date":"2020-08-17","arxiv_id":"2008.07079","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-adaptive-synchronization-approach-for","title":"An adaptive synchronization approach for weights of deep reinforcement learning","date":"2020-08-16","arxiv_id":"2008.06973","repositories_listed":0,"syntology":null},{"url":null,"slug":"drl-based-qos-aware-resource-allocation","title":"DRL-Based QoS-Aware Resource Allocation Scheme for Coexistence of Licensed and Unlicensed Users in LTE and Beyond","date":"2020-08-16","arxiv_id":"2008.06905","repositories_listed":0,"syntology":null},{"url":null,"slug":"inverse-reinforcement-learning-with-natural","title":"Inverse Reinforcement Learning with Natural Language Goals","date":"2020-08-16","arxiv_id":"2008.06924","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-braking-and-throttle-system-a-deep","title":"Autonomous Braking and Throttle System: A Deep Reinforcement Learning Approach for Naturalistic Driving","date":"2020-08-15","arxiv_id":"2008.06696","repositories_listed":0,"syntology":null},{"url":null,"slug":"chrome-dino-run-using-reinforcement-learning","title":"Chrome Dino Run using Reinforcement Learning","date":"2020-08-15","arxiv_id":"2008.06799","repositories_listed":0,"syntology":null},{"url":null,"slug":"explainability-in-deep-reinforcement-learning","title":"Explainability in Deep Reinforcement Learning","date":"2020-08-15","arxiv_id":"2008.06693","repositories_listed":0,"syntology":null},{"url":null,"slug":"decision-making-at-unsignalized-intersection","title":"Decision-making at Unsignalized Intersection for Autonomous Vehicles: Left-turn Maneuver with Deep Reinforcement Learning","date":"2020-08-14","arxiv_id":"2008.06595","repositories_listed":0,"syntology":null},{"url":null,"slug":"defending-adversarial-attacks-without","title":"Adversary Agnostic Robust Deep Reinforcement Learning","date":"2020-08-14","arxiv_id":"2008.06199","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-free-optimal-control-of-linear-multi","title":"Model-Free Optimal Control of Linear Multi-Agent Systems via Decomposition and Hierarchical Approximation","date":"2020-08-14","arxiv_id":"2008.06604","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-deep-reinforcement-learning-1","title":"Multi-Agent Deep Reinforcement Learning enabled Computation Resource Allocation in a Vehicular Cloud Network","date":"2020-08-14","arxiv_id":"2008.06464","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-trajectory","title":"Reinforcement Learning with Trajectory Feedback","date":"2020-08-13","arxiv_id":"2008.06036","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-image-matching-by-dynamic-feature","title":"Robust Image Matching By Dynamic Feature Selection","date":"2020-08-13","arxiv_id":"2008.05708","repositories_listed":0,"syntology":null},{"url":null,"slug":"visuomotor-mechanical-search-learning-to","title":"Visuomotor Mechanical Search: Learning to Retrieve Target Objects in Clutter","date":"2020-08-13","arxiv_id":"2008.06073","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-ocular-biomechanics-environment-for","title":"An ocular biomechanics environment for reinforcement learning","date":"2020-08-12","arxiv_id":"2008.05088","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-smart-1","title":"A Review of Deep Reinforcement Learning for Smart Building Energy Management","date":"2020-08-12","arxiv_id":"2008.05074","repositories_listed":0,"syntology":null}],"record_sha256":"be2440a507b9e69ae92e9d1caf16beb43c13ecaa9f24166c259fca91e6d8c059","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}