{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/deep-reinforcement-learning/papers/45","list_of":"/task/deep-reinforcement-learning","task":"Deep Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":45,"pages_in_order":59,"rows_per_page":100,"rows":[4401,4500],"of":5822,"counts":{"archive_papers_tagged":5822,"with_a_code_link":1739,"where_syntology_ran_a_sample":398,"not_listed_spam_title":0,"listed":5822,"listed_where_code_ran":398,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":340,"every_run_a_failure_of_syntologys_instrument":58,"listed_with_a_run_with_no_instrument_failure":340,"listed_every_run_a_failure_of_syntologys_instrument":58,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/deep-reinforcement-learning","prev":"/task/deep-reinforcement-learning/papers/44","next":"/task/deep-reinforcement-learning/papers/46","papers":[{"url":null,"slug":"a-state-representation-dueling-network-for","title":"A State Representation Dueling Network for Deep Reinforcement Learning","date":"2020-12-24","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"auto-agent-distiller-towards-efficient-deep","title":"Auto-Agent-Distiller: Towards Efficient Deep Reinforcement Learning Agents via Neural Architecture Search","date":"2020-12-24","arxiv_id":"2012.13091","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-vehicle-routing-problems-using","title":"Learning Vehicle Routing Problems using Policy Optimisation","date":"2020-12-24","arxiv_id":"2012.13269","repositories_listed":0,"syntology":null},{"url":null,"slug":"scc-an-efficient-deep-reinforcement-learning","title":"SCC: an efficient deep reinforcement learning agent mastering the game of StarCraft II","date":"2020-12-24","arxiv_id":"2012.13169","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethink-ai-based-power-grid-control-diving","title":"Rethink AI-based Power Grid Control: Diving Into Algorithm Design","date":"2020-12-23","arxiv_id":"2012.13026","repositories_listed":0,"syntology":null},{"url":null,"slug":"group-aware-robot-navigation-in-crowded","title":"Learning a Group-Aware Policy for Robot Navigation","date":"2020-12-22","arxiv_id":"2012.12291","repositories_listed":0,"syntology":null},{"url":null,"slug":"intelligent-resource-allocation-in-dense-lora","title":"Intelligent Resource Allocation in Dense LoRa Networks using Deep Reinforcement Learning","date":"2020-12-22","arxiv_id":"2012.11867","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-deep-reinforcement-learning-for","title":"Scalable Deep Reinforcement Learning for Routing and Spectrum Access in Physical Layer","date":"2020-12-22","arxiv_id":"2012.11783","repositories_listed":0,"syntology":null},{"url":null,"slug":"mobile-robot-planner-with-low-cost-cameras","title":"Mobile Robot Planner with Low-cost Cameras Using Deep Reinforcement Learning","date":"2020-12-21","arxiv_id":"2012.11160","repositories_listed":0,"syntology":null},{"url":null,"slug":"forming-human-robot-cooperation-for-tasks","title":"Forming Real-World Human-Robot Cooperation for Tasks With General Goal","date":"2020-12-19","arxiv_id":"2012.10773","repositories_listed":0,"syntology":null},{"url":null,"slug":"minimax-strikes-back","title":"Minimax Strikes Back","date":"2020-12-19","arxiv_id":"2012.10700","repositories_listed":0,"syntology":null},{"url":null,"slug":"embodied-visual-active-learning-for-semantic","title":"Embodied Visual Active Learning for Semantic Segmentation","date":"2020-12-17","arxiv_id":"2012.09503","repositories_listed":0,"syntology":null},{"url":null,"slug":"intrinsically-motivated-goal-conditioned","title":"Autotelic Agents with Intrinsically Motivated Goal-Conditioned Reinforcement Learning: a Short Survey","date":"2020-12-17","arxiv_id":"2012.09830","repositories_listed":0,"syntology":null},{"url":null,"slug":"magnet-multi-agent-graph-network-for-deep","title":"MAGNet: Multi-agent Graph Network for Deep Multi-agent Reinforcement Learning","date":"2020-12-17","arxiv_id":"2012.09762","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-optimal-district-heating-temperature","title":"Towards Optimal District Heating Temperature Control in China with Deep Reinforcement Learning","date":"2020-12-17","arxiv_id":"2012.09508","repositories_listed":0,"syntology":null},{"url":null,"slug":"batch-constrained-distributional","title":"Batch-Constrained Distributional Reinforcement Learning for Session-based Recommendation","date":"2020-12-16","arxiv_id":"2012.08984","repositories_listed":0,"syntology":null},{"url":null,"slug":"livemap-real-time-dynamic-map-in-automotive","title":"LiveMap: Real-Time Dynamic Map in Automotive Edge Computing","date":"2020-12-16","arxiv_id":"2012.10252","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-formulation-of-the","title":"A Reinforcement Learning Formulation of the Lyapunov Optimization: Application to Edge Computing Systems with Queue Stability","date":"2020-12-14","arxiv_id":"2012.07279","repositories_listed":0,"syntology":null},{"url":null,"slug":"ipm-move-planner-an-efficient-exploiting-deep","title":"IPM Move Planner: AN EFFICIENT EXPLOITING DEEP REINFORCEMENT LEARNING WITH MONTE CARLO TREE SEARCH","date":"2020-12-14","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learn-to-play-tetris-with-deep-reinforcement","title":"Learn to Play Tetris with Deep Reinforcement Learning","date":"2020-12-14","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-mobile-robot-navigation-in-the-dense","title":"Learning Mobile Robot Navigation in the Dense Crowd with Deep Reinforcement Learning","date":"2020-12-14","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"mobile-robots-exploration-via-deep","title":"Mobile Robots Exploration via Deep Reinforcement Learning","date":"2020-12-14","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"smoothing-deep-reinforcement-learning-for","title":"Smoothing Deep Reinforcement Learning for Power Control for Spectrum Sharing in Cognitive Radios","date":"2020-12-14","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-understanding-deep-policy-gradients-a","title":"Towards Understanding Deep Policy Gradients: A Case Study on PPO","date":"2020-12-14","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"regularizing-action-policies-for-smooth","title":"Regularizing Action Policies for Smooth Control with Reinforcement Learning","date":"2020-12-11","arxiv_id":"2012.06644","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-approach-for-4","title":"A Deep Reinforcement Learning Approach for Ramp Metering Based on Traffic Video Data","date":"2020-12-09","arxiv_id":"2012.12104","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-long-term","title":"Deep Reinforcement Learning for Long Term Hydropower Production Scheduling","date":"2020-12-09","arxiv_id":"2012.06312","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-stock","title":"Deep Reinforcement Learning for Stock Portfolio Optimization","date":"2020-12-09","arxiv_id":"2012.06325","repositories_listed":0,"syntology":null},{"url":null,"slug":"interactive-search-based-on-deep","title":"Interactive Search Based on Deep Reinforcement Learning","date":"2020-12-09","arxiv_id":"2012.06052","repositories_listed":0,"syntology":null},{"url":null,"slug":"emergence-of-different-modes-of-tool-use-in-a","title":"Emergence of Different Modes of Tool Use in a Reaching and Dragging Task","date":"2020-12-08","arxiv_id":"2012.04700","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-architectural-implications-of-distributed","title":"The Architectural Implications of Distributed Reinforcement Learning on CPU-GPU Systems","date":"2020-12-08","arxiv_id":"2012.04210","repositories_listed":0,"syntology":null},{"url":null,"slug":"battery-model-calibration-with-deep","title":"Battery Model Calibration with Deep Reinforcement Learning","date":"2020-12-07","arxiv_id":"2012.04010","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-policy-networks-for-npc-behaviors-that","title":"Deep Policy Networks for NPC Behaviors that Adapt to Changing Design Parameters in Roguelike Games","date":"2020-12-07","arxiv_id":"2012.03532","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-reservoir-management-through-deep","title":"Efficient Reservoir Management through Deep Reinforcement Learning","date":"2020-12-07","arxiv_id":"2012.03822","repositories_listed":0,"syntology":null},{"url":null,"slug":"fever-basketball-a-complex-flexible-and","title":"Fever Basketball: A Complex, Flexible, and Asynchronized Sports Game Environment for Multi-agent Reinforcement Learning","date":"2020-12-06","arxiv_id":"2012.03204","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-navigation-based-on-deep","title":"Multi-agent navigation based on deep reinforcement learning and traditional pathfinding algorithm","date":"2020-12-05","arxiv_id":"2012.09134","repositories_listed":0,"syntology":null},{"url":null,"slug":"demonstration-efficient-inverse-reinforcement","title":"Demonstration-efficient Inverse Reinforcement Learning in Procedurally Generated Environments","date":"2020-12-04","arxiv_id":"2012.02527","repositories_listed":0,"syntology":null},{"url":null,"slug":"playing-text-based-games-with-common-sense","title":"Playing Text-Based Games with Common Sense","date":"2020-12-04","arxiv_id":"2012.02757","repositories_listed":0,"syntology":null},{"url":null,"slug":"deepcrawl-deep-reinforcement-learning-for","title":"DeepCrawl: Deep Reinforcement Learning for Turn-based Strategy Games","date":"2020-12-03","arxiv_id":"2012.01914","repositories_listed":0,"syntology":null},{"url":null,"slug":"designing-a-prospective-covid-19-therapeutic","title":"Designing a Prospective COVID-19 Therapeutic with Reinforcement Learning","date":"2020-12-03","arxiv_id":"2012.01736","repositories_listed":0,"syntology":null},{"url":null,"slug":"partially-connected-automated-vehicle","title":"Partially Connected Automated Vehicle Cooperative Control Strategy with a Deep Reinforcement Learning Approach","date":"2020-12-03","arxiv_id":"2012.01841","repositories_listed":0,"syntology":null},{"url":null,"slug":"are-gradient-based-saliency-maps-useful-in","title":"Are Gradient-based Saliency Maps Useful in Deep Reinforcement Learning?","date":"2020-12-02","arxiv_id":"2012.01281","repositories_listed":0,"syntology":null},{"url":null,"slug":"coinbot-intelligent-robotic-coin-bag","title":"Coinbot: Intelligent Robotic Coin Bag Manipulation Using Deep Reinforcement Learning And Machine Teaching","date":"2020-12-02","arxiv_id":"2012.01356","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-learning-exploring-method-to-generate","title":"A Learning-Exploring Method to Generate Diverse Paraphrases with Multi-Objective Deep Reinforcement Learning","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"assessing-and-accelerating-coverage-in-deep","title":"Assessing and Accelerating Coverage in Deep Reinforcement Learning","date":"2020-12-01","arxiv_id":"2012.00724","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-temporal-difference-and-q-learning-learn-1","title":"Can Temporal-Diﬀerence and Q-Learning Learn Representation? A Mean-Field Theory","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"ecolight-intersection-control-in-developing","title":"EcoLight: Intersection Control in Developing Regions Under Extreme Budget and Network Constraints","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"instance-based-generalization-in-1","title":"Instance-based Generalization in Reinforcement Learning","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"storage-efficient-and-dynamic-flexible","title":"Storage Efficient and Dynamic Flexible Runtime Channel Pruning via Deep Reinforcement Learning","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-a-particle","title":"Deep reinforcement learning with a particle dynamics environment applied to emergency evacuation of a room with obstacles","date":"2020-11-30","arxiv_id":"2012.00065","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-controlled-learning-of-mdp","title":"Deep Controlled Learning for Inventory Control","date":"2020-11-30","arxiv_id":"2011.15122","repositories_listed":0,"syntology":null},{"url":null,"slug":"unicon-universal-neural-controller-for","title":"UniCon: Universal Neural Controller For Physics-based Character Motion","date":"2020-11-30","arxiv_id":"2011.15119","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-crowdsourced","title":"Deep Reinforcement Learning for Crowdsourced Urban Delivery: System States Characterization, Heuristics-guided Action Choice, and Rule-Interposing Integration","date":"2020-11-29","arxiv_id":"2011.14430","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptable-automation-with-modular-deep","title":"Adaptable Automation with Modular Deep Reinforcement Learning and Policy Transfer","date":"2020-11-27","arxiv_id":"2012.01934","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-wireless-1","title":"Deep Reinforcement Learning for Resource Constrained Multiclass Scheduling in Wireless Networks","date":"2020-11-27","arxiv_id":"2011.13634","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-time-active-vision-for-a-humanoid-soccer","title":"Real-time Active Vision for a Humanoid Soccer Robot Using Deep Reinforcement Learning","date":"2020-11-27","arxiv_id":"2011.13851","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-grid-topology-reconfiguration-using","title":"Exploring grid topology reconfiguration using a simple deep reinforcement learning approach","date":"2020-11-26","arxiv_id":"2011.13465","repositories_listed":0,"syntology":null},{"url":null,"slug":"predictive-per-balancing-priority-and","title":"Predictive PER: Balancing Priority and Diversity towards Stable Deep Reinforcement Learning","date":"2020-11-26","arxiv_id":"2011.13093","repositories_listed":0,"syntology":null},{"url":null,"slug":"metasensing-intelligent-metasurface-assisted","title":"MetaSensing: Intelligent Metasurface Assisted RF 3D Sensing by Deep Reinforcement Learning","date":"2020-11-25","arxiv_id":"2011.12515","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-playing-full-moba-games-with-deep-1","title":"Towards Playing Full MOBA Games with Deep Reinforcement Learning","date":"2020-11-25","arxiv_id":"2011.12692","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reusable-framework-based-on-reinforcement","title":"A Reusable Framework Based on Reinforcement Learning to Design Antennas for Curved Surfaces","date":"2020-11-24","arxiv_id":"2011.12131","repositories_listed":0,"syntology":null},{"url":null,"slug":"achieving-sample-efficient-and-online","title":"Learning of Long-Horizon Sparse-Reward Robotic Manipulator Tasks with Base Controllers","date":"2020-11-24","arxiv_id":"2011.12105","repositories_listed":0,"syntology":null},{"url":null,"slug":"powernet-multi-agent-deep-reinforcement","title":"PowerNet: Multi-agent Deep Reinforcement Learning for Scalable Powergrid Control","date":"2020-11-24","arxiv_id":"2011.12354","repositories_listed":0,"syntology":null},{"url":null,"slug":"repaint-knowledge-transfer-in-deep-actor-1","title":"REPAINT: Knowledge Transfer in Deep Reinforcement Learning","date":"2020-11-24","arxiv_id":"2011.11827","repositories_listed":0,"syntology":null},{"url":null,"slug":"cocoi-contact-aware-online-context-inference","title":"COCOI: Contact-aware Online Context Inference for Generalizable Non-planar Pushing","date":"2020-11-23","arxiv_id":"2011.11270","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-deep-reinforcement-learning-an","title":"Distributed Deep Reinforcement Learning: An Overview","date":"2020-11-22","arxiv_id":"2011.11012","repositories_listed":0,"syntology":null},{"url":null,"slug":"delay-constrained-buffer-aided-relay","title":"Delay Constrained Buffer-Aided Relay Selection in the Internet of Things with Decision-Assisted Reinforcement Learning","date":"2020-11-20","arxiv_id":"2011.10524","repositories_listed":0,"syntology":null},{"url":null,"slug":"energy-aware-deep-reinforcement-learning","title":"Energy Aware Deep Reinforcement Learning Scheduling for Sensors Correlated in Time and Space","date":"2020-11-19","arxiv_id":"2011.09747","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-model-selection-for-reinforcement","title":"Online Model Selection for Reinforcement Learning with Function Approximation","date":"2020-11-19","arxiv_id":"2011.09750","repositories_listed":0,"syntology":null},{"url":null,"slug":"indoor-point-to-point-navigation-with-deep","title":"Indoor Point-to-Point Navigation with Deep Reinforcement Learning and Ultra-wideband","date":"2020-11-18","arxiv_id":"2011.09241","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-and-permissioned","title":"Deep Reinforcement Learning and Permissioned Blockchain for Content Caching in Vehicular Edge Computing and Networks","date":"2020-11-17","arxiv_id":"2011.08449","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-stochastic","title":"Deep Reinforcement Learning for Stochastic Computation Offloading in Digital Twin Networks","date":"2020-11-17","arxiv_id":"2011.08430","repositories_listed":0,"syntology":null},{"url":null,"slug":"edge-intelligence-for-energy-efficient","title":"Edge Intelligence for Energy-efficient Computation Offloading and Resource Allocation in 5G Beyond","date":"2020-11-17","arxiv_id":"2011.08442","repositories_listed":0,"syntology":null},{"url":null,"slug":"passgoodpool-joint-passengers-and-goods-fleet","title":"PassGoodPool: Joint Passengers and Goods Fleet Management with Reinforcement Learning aided Pricing, Matching, and Route Planning","date":"2020-11-17","arxiv_id":"2011.08999","repositories_listed":0,"syntology":null},{"url":null,"slug":"time-efficient-mars-exploration-of","title":"Time-Efficient Mars Exploration of Simultaneous Coverage and Charging with Multiple Drones","date":"2020-11-16","arxiv_id":"2011.07759","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-human-level-learning-of-complex","title":"Data-Efficient Learning for Complex and Real-Time Physical Problem Solving using Augmented Simulation","date":"2020-11-14","arxiv_id":"2011.07193","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-of-transition","title":"Deep Reinforcement Learning of Transition States","date":"2020-11-13","arxiv_id":"2011.06700","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-quadruped-jumping-via-deep","title":"Robust Quadruped Jumping via Deep Reinforcement Learning","date":"2020-11-13","arxiv_id":"2011.07089","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-layer-optimization-and-distributed","title":"Cross Layer Optimization and Distributed Reinforcement Learning for Wireless 360° Video Streaming","date":"2020-11-12","arxiv_id":"2011.06356","repositories_listed":0,"syntology":null},{"url":null,"slug":"domain-level-explainability-a-challenge-for","title":"Domain-Level Explainability -- A Challenge for Creating Trust in Superhuman AI Strategies","date":"2020-11-12","arxiv_id":"2011.06665","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-neural-architectures-for-recommender","title":"Adaptive Neural Architectures for Recommender Systems","date":"2020-11-11","arxiv_id":"2012.00743","repositories_listed":0,"syntology":null},{"url":null,"slug":"behaviorally-diverse-traffic-simulation-via","title":"Behaviorally Diverse Traffic Simulation via Reinforcement Learning","date":"2020-11-11","arxiv_id":"2011.05741","repositories_listed":0,"syntology":null},{"url":null,"slug":"proximal-policy-optimization-via-enhanced","title":"Proximal Policy Optimization via Enhanced Exploration Efficiency","date":"2020-11-11","arxiv_id":"2011.05525","repositories_listed":0,"syntology":null},{"url":null,"slug":"sim-to-real-transfer-for-miniature-autonomous","title":"Sim-To-Real Transfer for Miniature Autonomous Car Racing","date":"2020-11-11","arxiv_id":"2011.05617","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-relay-selection-and-power-allocation","title":"Hierarchical Reinforcement Learning for Relay Selection and Power Optimization in Two-Hop Cooperative Relay Network","date":"2020-11-10","arxiv_id":"2011.04891","repositories_listed":0,"syntology":null},{"url":null,"slug":"perturbation-based-exploration-methods-in","title":"Perturbation-based exploration methods in deep reinforcement learning","date":"2020-11-10","arxiv_id":"2011.05446","repositories_listed":0,"syntology":null},{"url":null,"slug":"challenges-of-applying-deep-reinforcement","title":"Challenges of Applying Deep Reinforcement Learning in Dynamic Dispatching","date":"2020-11-09","arxiv_id":"2011.05570","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-navigation-in","title":"Deep Reinforcement Learning for Navigation in AAA Video Games","date":"2020-11-09","arxiv_id":"2011.04764","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-ran","title":"Deep reinforcement learning for RAN optimization and control","date":"2020-11-09","arxiv_id":"2011.04607","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-age-of-information-through-aerial","title":"Optimizing Age of Information Through Aerial Reconfigurable Intelligent Surfaces: A Deep Reinforcement Learning Approach","date":"2020-11-09","arxiv_id":"2011.04817","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-autonomous-driving","title":"Reinforcement Learning for Autonomous Driving with Latent State Inference and Spatial-Temporal Relationships","date":"2020-11-09","arxiv_id":"2011.04251","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-market-power-using-deep","title":"Exploring market power using deep reinforcement learning for intelligent bidding strategies","date":"2020-11-08","arxiv_id":"2011.04079","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-role-of-planning-in-model-based-deep-1","title":"On the role of planning in model-based deep reinforcement learning","date":"2020-11-08","arxiv_id":"2011.04021","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-few-shot-adaptation-of-visual-navigation","title":"A Few Shot Adaptation of Visual Navigation Skills to New Observations using Meta-Learning","date":"2020-11-06","arxiv_id":"2011.03609","repositories_listed":0,"syntology":null},{"url":null,"slug":"motion-prediction-on-self-driving-cars-a","title":"Motion Prediction on Self-driving Cars: A Review","date":"2020-11-06","arxiv_id":"2011.03635","repositories_listed":0,"syntology":null},{"url":null,"slug":"single-and-multi-agent-deep-reinforcement","title":"Single and Multi-Agent Deep Reinforcement Learning for AI-Enabled Wireless Networks: A Tutorial","date":"2020-11-06","arxiv_id":"2011.03615","repositories_listed":0,"syntology":null},{"url":null,"slug":"lbgp-learning-based-goal-planning-for","title":"LBGP: Learning Based Goal Planning for Autonomous Following in Front","date":"2020-11-05","arxiv_id":"2011.03125","repositories_listed":0,"syntology":null},{"url":null,"slug":"playing-optical-tweezers-with-deep","title":"Playing optical tweezers with deep reinforcement learning: in virtual, physical and augmented environments","date":"2020-11-05","arxiv_id":"2011.04424","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-inverse-deep-reinforcement","title":"Generative Inverse Deep Reinforcement Learning for Online Recommendation","date":"2020-11-04","arxiv_id":"2011.02248","repositories_listed":0,"syntology":null},{"url":null,"slug":"mbvi-model-based-value-initialization-for","title":"Optimal Control-Based Baseline for Guided Exploration in Policy Gradient Methods","date":"2020-11-04","arxiv_id":"2011.02073","repositories_listed":0,"syntology":null}],"record_sha256":"64562ba882360f049a996658d2c318eb876eaeabefd5cc9d3f8a8fd36c796ce1","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}