{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/deep-reinforcement-learning/papers/49","list_of":"/task/deep-reinforcement-learning","task":"Deep Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":49,"pages_in_order":59,"rows_per_page":100,"rows":[4801,4900],"of":5822,"counts":{"archive_papers_tagged":5822,"with_a_code_link":1739,"where_syntology_ran_a_sample":398,"not_listed_spam_title":0,"listed":5822,"listed_where_code_ran":398,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":340,"every_run_a_failure_of_syntologys_instrument":58,"listed_with_a_run_with_no_instrument_failure":340,"listed_every_run_a_failure_of_syntologys_instrument":58,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/deep-reinforcement-learning","prev":"/task/deep-reinforcement-learning/papers/48","next":"/task/deep-reinforcement-learning/papers/50","papers":[{"url":null,"slug":"a-deep-reinforcement-learning-approach-to-2","title":"A Deep Reinforcement Learning Approach to Efficient Drone Mobility Support","date":"2020-05-11","arxiv_id":"2005.05229","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-organ","title":"Deep Reinforcement Learning for Organ Localization in CT","date":"2020-05-11","arxiv_id":"2005.04974","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-pid-and-antiwindup-control-design-as","title":"Optimal PID and Antiwindup Control Design as a Reinforcement Learning Problem","date":"2020-05-10","arxiv_id":"2005.04539","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-deep-reinforcement-learning-ready-for","title":"Is Deep Reinforcement Learning Ready for Practical Applications in Healthcare? A Sensitivity Analysis of Duel-DDQN for Hemodynamic Management in Sepsis Patients","date":"2020-05-08","arxiv_id":"2005.04301","repositories_listed":0,"syntology":null},{"url":null,"slug":"robotic-arm-control-and-task-training-through","title":"Robotic Arm Control and Task Training through Deep Reinforcement Learning","date":"2020-05-06","arxiv_id":"2005.02632","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-through-meta","title":"Safe Reinforcement Learning through Meta-learned Instincts","date":"2020-05-06","arxiv_id":"2005.03233","repositories_listed":0,"syntology":null},{"url":null,"slug":"demand-side-scheduling-based-on-deep-actor","title":"Demand-Side Scheduling Based on Multi-Agent Deep Actor-Critic Learning for Smart Grids","date":"2020-05-05","arxiv_id":"2005.01979","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalized-planning-with-deep-reinforcement","title":"Generalized Planning With Deep Reinforcement Learning","date":"2020-05-05","arxiv_id":"2005.02305","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-intelligent-1","title":"Deep Reinforcement Learning for Intelligent Transportation Systems: A Survey","date":"2020-05-02","arxiv_id":"2005.00935","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-beam-association-for-high-mobility","title":"Optimal Beam Association for High Mobility mmWave Vehicular Networks: Lightweight Parallel Reinforcement Learning Approach","date":"2020-05-02","arxiv_id":"2005.00694","repositories_listed":0,"syntology":null},{"url":null,"slug":"episodic-reinforcement-learning-with","title":"Episodic Reinforcement Learning with Associative Memory","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-heuristics-for-quantified-boolean","title":"Learning Heuristics for Quantified Boolean Formulas through Reinforcement Learning","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"synthesizing-programmatic-policies-that","title":"Synthesizing Programmatic Policies that Inductively Generalize","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-evaluating-robustness-of-deep","title":"Toward Evaluating Robustness of Deep Reinforcement Learning with Continuous Control","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"bootstrap-latent-predictive-representations","title":"Bootstrap Latent-Predictive Representations for Multitask Reinforcement Learning","date":"2020-04-30","arxiv_id":"2004.14646","repositories_listed":0,"syntology":null},{"url":null,"slug":"breaking-global-barriers-in-parallel","title":"Breaking (Global) Barriers in Parallel Stochastic Optimization with Wait-Avoiding Group Averaging","date":"2020-04-30","arxiv_id":"2005.00124","repositories_listed":0,"syntology":null},{"url":null,"slug":"sim-to-real-transfer-with-incremental","title":"Sim-to-Real Transfer with Incremental Environment Complexity for Reinforcement Learning of Depth-Based Robot Navigation","date":"2020-04-30","arxiv_id":"2004.14684","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-target-driven-visual-navigation","title":"Improving Target-driven Visual Navigation with Attention on 3D Spatial Relationships","date":"2020-04-29","arxiv_id":"2005.02153","repositories_listed":0,"syntology":null},{"url":null,"slug":"molecular-design-in-synthetically-accessible","title":"Molecular Design in Synthetically Accessible Chemical Space via Deep Reinforcement Learning","date":"2020-04-29","arxiv_id":"2004.14308","repositories_listed":0,"syntology":null},{"url":null,"slug":"warm-start-alphazero-self-play-search","title":"Warm-Start AlphaZero Self-Play Search Enhancements","date":"2020-04-26","arxiv_id":"2004.12357","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-state-aggregation-approach-for-solving","title":"A State Aggregation Approach for Solving Knapsack Problem with Deep Reinforcement Learning","date":"2020-04-25","arxiv_id":"2004.12117","repositories_listed":0,"syntology":null},{"url":null,"slug":"cooperative-perception-with-deep","title":"Cooperative Perception with Deep Reinforcement Learning for Connected Vehicles","date":"2020-04-23","arxiv_id":"2004.10927","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-dialog-policies-from-weak","title":"Learning Dialog Policies from Weak Demonstrations","date":"2020-04-23","arxiv_id":"2004.11054","repositories_listed":0,"syntology":null},{"url":null,"slug":"autoeg-automated-experience-grafting-for-off","title":"AutoEG: Automated Experience Grafting for Off-Policy Deep Reinforcement Learning","date":"2020-04-22","arxiv_id":"2004.10698","repositories_listed":0,"syntology":null},{"url":null,"slug":"flexible-and-efficient-long-range-planning-1","title":"Flexible and Efficient Long-Range Planning Through Curious Exploration","date":"2020-04-22","arxiv_id":"2004.10876","repositories_listed":0,"syntology":null},{"url":null,"slug":"qd-tree-learning-data-layouts-for-big-data","title":"Qd-tree: Learning Data Layouts for Big Data Analytics","date":"2020-04-22","arxiv_id":"2004.10898","repositories_listed":0,"syntology":null},{"url":null,"slug":"stdpg-a-spatio-temporal-deterministic-policy","title":"STDPG: A Spatio-Temporal Deterministic Policy Gradient Agent for Dynamic Routing in SDN","date":"2020-04-21","arxiv_id":"2004.09783","repositories_listed":0,"syntology":null},{"url":null,"slug":"approximate-exploitability-learning-a-best","title":"Approximate exploitability: Learning a best response in large games","date":"2020-04-20","arxiv_id":"2004.09677","repositories_listed":0,"syntology":null},{"url":null,"slug":"macro-action-based-deep-multi-agent","title":"Macro-Action-Based Deep Multi-Agent Reinforcement Learning","date":"2020-04-18","arxiv_id":"2004.08646","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-adaptive-1","title":"Deep Reinforcement Learning for Adaptive Learning Systems","date":"2020-04-17","arxiv_id":"2004.08410","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-guided-deep-reinforcement-learning","title":"Knowledge-guided Deep Reinforcement Learning for Interactive Recommendation","date":"2020-04-17","arxiv_id":"2004.08068","repositories_listed":0,"syntology":null},{"url":null,"slug":"actionspotter-deep-reinforcement-learning","title":"ActionSpotter: Deep Reinforcement Learning Framework for Temporal Action Spotting in Videos","date":"2020-04-15","arxiv_id":"2004.06971","repositories_listed":0,"syntology":null},{"url":null,"slug":"extending-deep-reinforcement-learning","title":"Extending Deep Reinforcement Learning Frameworks in Cryptocurrency Market Making","date":"2020-04-15","arxiv_id":"2004.06985","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-user-pairing-and-association-for","title":"Joint User Pairing and Association for Multicell NOMA: A Pointer Network-based Approach","date":"2020-04-15","arxiv_id":"2004.07395","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-deep-reinforcement-learning-based","title":"Safe deep reinforcement learning-based constrained optimal control scheme for active distribution networks","date":"2020-04-15","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"actor-critic-deep-reinforcement-learning-for-1","title":"Actor-Critic Deep Reinforcement Learning for Solving Job Shop Scheduling Problems","date":"2020-04-14","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-evaluation-of-autonomous-vehicles","title":"Adversarial Evaluation of Autonomous Vehicles in Lane-Change Scenarios","date":"2020-04-14","arxiv_id":"2004.06531","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-framework-for-2","title":"A Deep Reinforcement Learning Framework for Continuous Intraday Market Bidding","date":"2020-04-13","arxiv_id":"2004.05940","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-non-cooperative-meta-modeling-game-for","title":"A non-cooperative meta-modeling game for automated third-party calibrating, validating, and falsifying constitutive laws with parallelized adversarial attacks","date":"2020-04-13","arxiv_id":"2004.09392","repositories_listed":0,"syntology":null},{"url":null,"slug":"thinking-while-moving-deep-reinforcement-1","title":"Thinking While Moving: Deep Reinforcement Learning with Concurrent Control","date":"2020-04-13","arxiv_id":"2004.06089","repositories_listed":0,"syntology":null},{"url":null,"slug":"certified-adversarial-robustness-for-deep-1","title":"Certifiable Robustness to Adversarial State Uncertainty in Deep Reinforcement Learning","date":"2020-04-11","arxiv_id":"2004.06496","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-process-1","title":"Deep Reinforcement Learning for Process Control: A Primer for Beginners","date":"2020-04-11","arxiv_id":"2004.05490","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-drl-another","title":"Deep Reinforcement Learning (DRL): Another Perspective for Unsupervised Wireless Localization","date":"2020-04-09","arxiv_id":"2004.04618","repositories_listed":0,"syntology":null},{"url":null,"slug":"resource-management-for-blockchain-enabled","title":"Resource Management for Blockchain-enabled Federated Learning: A Deep Reinforcement Learning Approach","date":"2020-04-08","arxiv_id":"2004.04104","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-do-you-act-an-empirical-study-to","title":"How Do You Act? An Empirical Study to Understand Behavior of Deep Reinforcement Learning Agents","date":"2020-04-07","arxiv_id":"2004.03237","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-generative-adversarial-nets-on-atari","title":"Using Generative Adversarial Nets on Atari Games for Feature Extraction in Deep Reinforcement Learning","date":"2020-04-06","arxiv_id":"2004.02762","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantification-of-tomographic-patterns","title":"Automated Quantification of CT Patterns Associated with COVID-19 from Chest CT","date":"2020-04-02","arxiv_id":"2004.01279","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-reference-reinforcement-learning","title":"Model-Reference Reinforcement Learning Control of Autonomous Surface Vehicles with Uncertainties","date":"2020-03-30","arxiv_id":"2003.13839","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-based-multi-channel-access-in-5g-and","title":"Learning-Based Multi-Channel Access in 5G and Beyond Networks with Fast Time-Varying Channels","date":"2020-03-28","arxiv_id":"2003.14403","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-medical-triage-from-clinicians-using","title":"Learning medical triage from clinicians using Deep Q-Learning","date":"2020-03-28","arxiv_id":"2003.12828","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-better-opioid-antagonists-using-deep","title":"Towards Better Opioid Antagonists Using Deep Reinforcement Learning","date":"2020-03-26","arxiv_id":"2004.04768","repositories_listed":0,"syntology":null},{"url":null,"slug":"driver-modeling-through-deep-reinforcement","title":"Driver Modeling through Deep Reinforcement Learning and Behavioral Game Theory","date":"2020-03-24","arxiv_id":"2003.11071","repositories_listed":0,"syntology":null},{"url":null,"slug":"learn-to-schedule-leasch-a-deep-reinforcement","title":"Learn to Schedule (LEASCH): A Deep reinforcement learning approach for radio resource scheduling in the 5G MAC layer","date":"2020-03-24","arxiv_id":"2003.11003","repositories_listed":0,"syntology":null},{"url":null,"slug":"do-recent-advancements-in-model-based-deep-1","title":"Importance of using appropriate baselines for evaluation of data-efficiency in deep reinforcement learning for Atari","date":"2020-03-23","arxiv_id":"2003.10181","repositories_listed":0,"syntology":null},{"url":null,"slug":"incorporating-relational-background-knowledge","title":"Incorporating Relational Background Knowledge into Reinforcement Learning via Differentiable Inductive Logic Programming","date":"2020-03-23","arxiv_id":"2003.10386","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-walk-spike-based-reinforcement","title":"Learning to Walk: Spike Based Reinforcement Learning for Hexapod Robot Central Pattern Generation","date":"2020-03-22","arxiv_id":"2003.10026","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-deep-reinforcement-learning-with","title":"Accelerating Deep Reinforcement Learning With the Aid of Partial Model: Energy-Efficient Predictive Video Streaming","date":"2020-03-21","arxiv_id":"2003.09708","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-uav-navigation-a-ddpg-based-deep","title":"Autonomous UAV Navigation: A DDPG-based Deep Reinforcement Learning Approach","date":"2020-03-21","arxiv_id":"2003.10923","repositories_listed":0,"syntology":null},{"url":null,"slug":"comprehensive-review-of-deep-reinforcement","title":"Comprehensive Review of Deep Reinforcement Learning Methods and Applications in Economics","date":"2020-03-21","arxiv_id":"2004.01509","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-smooth","title":"Deep Reinforcement Learning with Robust and Smooth Policy","date":"2020-03-21","arxiv_id":"2003.09534","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-weighted-q","title":"Deep Reinforcement Learning with Weighted Q-Learning","date":"2020-03-20","arxiv_id":"2003.09280","repositories_listed":0,"syntology":null},{"url":null,"slug":"exchangeable-input-representations-for","title":"Exchangeable Input Representations for Reinforcement Learning","date":"2020-03-19","arxiv_id":"2003.09022","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-cognitive-routing-based-on-deep","title":"Towards Cognitive Routing based on Deep Reinforcement Learning","date":"2020-03-19","arxiv_id":"2003.12439","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-socially-acceptable-perturbations","title":"Generating Socially Acceptable Perturbations for Efficient Evaluation of Autonomous Vehicles","date":"2020-03-18","arxiv_id":"2003.08034","repositories_listed":0,"syntology":null},{"url":null,"slug":"placement-optimization-with-deep","title":"Placement Optimization with Deep Reinforcement Learning","date":"2020-03-18","arxiv_id":"2003.08445","repositories_listed":0,"syntology":null},{"url":null,"slug":"viewport-aware-deep-reinforcement-learning","title":"Viewport-Aware Deep Reinforcement Learning Approach for 360$^o$ Video Caching","date":"2020-03-18","arxiv_id":"2003.08473","repositories_listed":0,"syntology":null},{"url":null,"slug":"watch-your-back-backdoor-attacks-in-deep","title":"Stop-and-Go: Exploring Backdoor Attacks on Deep Reinforcement Learning-based Traffic Congestion Control Systems","date":"2020-03-17","arxiv_id":"2003.07859","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-performance-in-reinforcement","title":"Improving Performance in Reinforcement Learning by Breaking Generalization in Neural Networks","date":"2020-03-16","arxiv_id":"2003.07417","repositories_listed":0,"syntology":null},{"url":null,"slug":"application-of-deep-q-network-in-portfolio","title":"Application of Deep Q-Network in Portfolio Management","date":"2020-03-13","arxiv_id":"2003.06365","repositories_listed":0,"syntology":null},{"url":null,"slug":"analyzing-visual-representations-in-embodied","title":"Analyzing Visual Representations in Embodied Navigation Tasks","date":"2020-03-12","arxiv_id":"2003.05993","repositories_listed":0,"syntology":null},{"url":null,"slug":"uav-to-device-underlay-communications-age-of","title":"UAV-to-Device Underlay Communications: Age of Information Minimization by Multi-agent Deep Reinforcement Learning","date":"2020-03-12","arxiv_id":"2003.05830","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-curriculum-learning-for-deep-rl-a","title":"Automatic Curriculum Learning For Deep RL: A Short Survey","date":"2020-03-10","arxiv_id":"2003.04664","repositories_listed":0,"syntology":null},{"url":null,"slug":"privacy-cost-management-in-smart-meters-using","title":"Privacy-Cost Management in Smart Meters Using Deep Reinforcement Learning","date":"2020-03-10","arxiv_id":"2003.04946","repositories_listed":0,"syntology":null},{"url":null,"slug":"squirl-robust-and-efficient-learning-from","title":"SQUIRL: Robust and Efficient Learning from Video Demonstration of Long-Horizon Robotic Manipulation Tasks","date":"2020-03-10","arxiv_id":"2003.04956","repositories_listed":0,"syntology":null},{"url":"/paper/the-minerl-competition-on-sample-efficient-1","slug":"the-minerl-competition-on-sample-efficient-1","title":"Retrospective Analysis of the 2019 MineRL Competition on Sample Efficient Reinforcement Learning","date":"2020-03-10","arxiv_id":"2003.05012","repositories_listed":0,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-minerl-competition-on-sample-efficient-1#ran","syntology_url":"https://syntology.ai/paper/2003.05012","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.05012"}},"official":null}},{"url":null,"slug":"efficiency-and-equity-are-both-essential-a","title":"Efficiency and Equity are Both Essential: A Generalized Traffic Signal Controller with Deep Reinforcement Learning","date":"2020-03-09","arxiv_id":"2003.04046","repositories_listed":0,"syntology":null},{"url":null,"slug":"cost-sensitive-portfolio-selection-via-deep","title":"Cost-Sensitive Portfolio Selection via Deep Reinforcement Learning","date":"2020-03-06","arxiv_id":"2003.03051","repositories_listed":0,"syntology":null},{"url":null,"slug":"balance-between-efficient-and-effective","title":"Balance Between Efficient and Effective Learning: Dense2Sparse Reward Shaping for Robot Manipulation with Environment Uncertainty","date":"2020-03-05","arxiv_id":"2003.02740","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-robust","title":"Deep Reinforcement Learning-BasedRobust Protection in DER-Rich Distribution Grids","date":"2020-03-05","arxiv_id":"2003.02422","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-and-effective-similar-subtrajectory","title":"Efficient and Effective Similar Subtrajectory Search with Deep Reinforcement Learning","date":"2020-03-05","arxiv_id":"2003.02542","repositories_listed":0,"syntology":null},{"url":null,"slug":"privacy-aware-time-series-data-sharing-with","title":"Privacy-Aware Time-Series Data Sharing with Deep Reinforcement Learning","date":"2020-03-04","arxiv_id":"2003.02685","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-qos","title":"Deep Reinforcement Learning for QoS-Constrained Resource Allocation in Multiservice Networks","date":"2020-03-03","arxiv_id":"2003.02643","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-object-level-deep","title":"Relevance-Guided Modeling of Object Dynamics for Reinforcement Learning","date":"2020-03-03","arxiv_id":"2003.01384","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-world-human-robot-collaborative","title":"Real-World Human-Robot Collaborative Reinforcement Learning","date":"2020-03-02","arxiv_id":"2003.01156","repositories_listed":0,"syntology":null},{"url":null,"slug":"v2i-connectivity-based-dynamic-queue-jumper","title":"Dynamic Queue-Jump Lane for Emergency Vehicles under Partially Connected Settings: A Multi-Agent Deep Reinforcement Learning Approach","date":"2020-03-02","arxiv_id":"2003.01025","repositories_listed":0,"syntology":null},{"url":null,"slug":"learn-task-first-or-learn-human-partner-first","title":"Learn Task First or Learn Human Partner First: A Hierarchical Task Decomposition Method for Human-Robot Cooperation","date":"2020-03-01","arxiv_id":"2003.00400","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-flipit","title":"Deep Reinforcement Learning for FlipIt Security Game","date":"2020-02-28","arxiv_id":"2002.12909","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-visual-communication-map-for-multi-agent","title":"A Visual Communication Map for Multi-Agent Deep Reinforcement Learning","date":"2020-02-27","arxiv_id":"2002.11882","repositories_listed":0,"syntology":null},{"url":null,"slug":"acceleration-of-actor-critic-deep","title":"Acceleration of Actor-Critic Deep Reinforcement Learning for Visual Grasping in Clutter by State Representation Learning Based on Disentanglement of a Raw Input Image","date":"2020-02-27","arxiv_id":"2002.11903","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-intelligent","title":"Deep Reinforcement Learning Based Intelligent Reflecting Surface for Secure Wireless Communications","date":"2020-02-27","arxiv_id":"2002.12271","repositories_listed":0,"syntology":null},{"url":null,"slug":"review-analyze-and-design-a-comprehensive","title":"Review, Analysis and Design of a Comprehensive Deep Reinforcement Learning Framework","date":"2020-02-27","arxiv_id":"2002.11883","repositories_listed":0,"syntology":null},{"url":null,"slug":"when-do-drivers-concentrate-attention-based","title":"When Do Drivers Concentrate? Attention-based Driver Behavior Modeling With Deep Reinforcement Learning","date":"2020-02-26","arxiv_id":"2002.11385","repositories_listed":0,"syntology":null},{"url":null,"slug":"metric-based-imitation-learning-between-two","title":"Metric-Based Imitation Learning Between Two Dissimilar Anthropomorphic Robotic Arms","date":"2020-02-25","arxiv_id":"2003.02638","repositories_listed":0,"syntology":null},{"url":null,"slug":"simultaneously-evolving-deep-reinforcement","title":"Simultaneously Evolving Deep Reinforcement Learning Models using Multifactorial Optimization","date":"2020-02-25","arxiv_id":"2002.12133","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-transferable-are-the-representations","title":"How Transferable are the Representations Learned by Deep Q Agents?","date":"2020-02-24","arxiv_id":"2002.10021","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-linear","title":"Deep Reinforcement Learning with Linear Quadratic Regulator Regions","date":"2020-02-23","arxiv_id":"2002.09820","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-traffic-lights-with-multi-agent","title":"Optimizing Traffic Lights with Multi-agent Deep Reinforcement Learning and V2X communication","date":"2020-02-23","arxiv_id":"2002.09853","repositories_listed":0,"syntology":null},{"url":null,"slug":"periodic-q-learning","title":"Periodic Q-Learning","date":"2020-02-23","arxiv_id":"2002.09795","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-data-augmentation-via-deep","title":"Automatic Data Augmentation via Deep Reinforcement Learning for Effective Kidney Tumor Segmentation","date":"2020-02-22","arxiv_id":"2002.09703","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-learning-for-ultra-reliable-and-low","title":"Deep Learning for Ultra-Reliable and Low-Latency Communications in 6G Networks","date":"2020-02-22","arxiv_id":"2002.11045","repositories_listed":0,"syntology":null}],"record_sha256":"f9e2354d26de0db9bcb25c1dfcb1c60f6827db50e3cdcba7f09ce3d9543ef732","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}