{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/deep-reinforcement-learning/papers/31","list_of":"/task/deep-reinforcement-learning","task":"Deep Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":31,"pages_in_order":59,"rows_per_page":100,"rows":[3001,3100],"of":5822,"counts":{"archive_papers_tagged":5822,"with_a_code_link":1739,"where_syntology_ran_a_sample":398,"not_listed_spam_title":0,"listed":5822,"listed_where_code_ran":398,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":340,"every_run_a_failure_of_syntologys_instrument":58,"listed_with_a_run_with_no_instrument_failure":340,"listed_every_run_a_failure_of_syntologys_instrument":58,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/deep-reinforcement-learning","prev":"/task/deep-reinforcement-learning/papers/30","next":"/task/deep-reinforcement-learning/papers/32","papers":[{"url":null,"slug":"value-function-estimation-using-conditional","title":"Value function estimation using conditional diffusion models for control","date":"2023-06-09","arxiv_id":"2306.07290","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-framework-for-dynamically-training-and","title":"A framework for dynamically training and adapting deep reinforcement learning models to different, low-compute, and continuously changing radiology deployment environments","date":"2023-06-08","arxiv_id":"2306.05310","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-newborn-embodied-turing-test-for-view","title":"A newborn embodied Turing test for view-invariant object recognition","date":"2023-06-08","arxiv_id":"2306.05582","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-navigate-in-turbulent-flows-with","title":"Learning to Navigate in Turbulent Flows with Aerial Robot Swarms: A Cooperative Deep Reinforcement Learning Approach","date":"2023-06-07","arxiv_id":"2306.04781","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-control-of-4","title":"Reinforcement Learning-Based Control of CrazyFlie 2.X Quadrotor","date":"2023-06-06","arxiv_id":"2306.03951","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-novel-multi-agent-deep-rl-approach-for","title":"A Novel Multi-Agent Deep RL Approach for Traffic Signal Control","date":"2023-06-05","arxiv_id":"2306.02684","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-q-learning-versus-proximal-policy","title":"Deep Q-Learning versus Proximal Policy Optimization: Performance Comparison in a Material Sorting Task","date":"2023-06-02","arxiv_id":"2306.01451","repositories_listed":0,"syntology":null},{"url":null,"slug":"dvfo-dynamic-voltage-frequency-scaling-and","title":"DVFO: Learning-Based DVFS for Energy-Efficient Edge-Cloud Collaborative Inference","date":"2023-06-02","arxiv_id":"2306.01811","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-the-generalizability-and-robustness","title":"Improving the generalizability and robustness of large-scale traffic signal control","date":"2023-06-02","arxiv_id":"2306.01925","repositories_listed":0,"syntology":null},{"url":null,"slug":"bite-accelerating-learned-query-optimization","title":"BitE : Accelerating Learned Query Optimization in a Mixed-Workload Environment","date":"2023-06-01","arxiv_id":"2306.00845","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-navigation-strategies-in-the","title":"Investigating Navigation Strategies in the Morris Water Maze through Deep Reinforcement Learning","date":"2023-06-01","arxiv_id":"2306.01066","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-environment-lifelong-deep-reinforcement","title":"Multi-environment lifelong deep reinforcement learning for medical imaging","date":"2023-05-31","arxiv_id":"2306.00188","repositories_listed":0,"syntology":null},{"url":null,"slug":"framm-fair-ranking-with-missing-modalities","title":"FRAMM: Fair Ranking with Missing Modalities for Clinical Trial Site Selection","date":"2023-05-30","arxiv_id":"2305.19407","repositories_listed":0,"syntology":null},{"url":null,"slug":"action-valuation-of-on-and-off-ball-soccer","title":"Action valuation of on- and off-ball soccer players based on multi-agent deep reinforcement learning","date":"2023-05-29","arxiv_id":"2305.17886","repositories_listed":0,"syntology":null},{"url":null,"slug":"doing-the-right-thing-for-the-right-reason","title":"Doing the right thing for the right reason: Evaluating artificial moral cognition by probing cost insensitivity","date":"2023-05-29","arxiv_id":"2305.18269","repositories_listed":0,"syntology":null},{"url":null,"slug":"perimeter-control-using-deep-reinforcement","title":"Perimeter Control Using Deep Reinforcement Learning: A Model-free Approach towards Homogeneous Flow Rate Optimization","date":"2023-05-29","arxiv_id":"2305.19291","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comprehensive-overview-and-comparative","title":"A Comprehensive Overview and Comparative Analysis on Deep Learning Models: CNN, RNN, LSTM, GRU","date":"2023-05-27","arxiv_id":"2305.17473","repositories_listed":0,"syntology":null},{"url":null,"slug":"lucy-skg-learning-to-play-rocket-league","title":"Lucy-SKG: Learning to Play Rocket League Efficiently Using Deep Reinforcement Learning","date":"2023-05-25","arxiv_id":"2305.15801","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-plasticity","title":"Deep Reinforcement Learning with Plasticity Injection","date":"2023-05-24","arxiv_id":"2305.15555","repositories_listed":0,"syntology":null},{"url":null,"slug":"successor-predecessor-intrinsic-exploration","title":"Successor-Predecessor Intrinsic Exploration","date":"2023-05-24","arxiv_id":"2305.15277","repositories_listed":0,"syntology":null},{"url":null,"slug":"control-of-a-simulated-mri-scanner-with-deep","title":"Control of a simulated MRI scanner with deep reinforcement learning","date":"2023-05-23","arxiv_id":"2305.13979","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-multi-1","title":"Deep Reinforcement Learning-based Multi-objective Path Planning on the Off-road Terrain Environment for Ground Vehicles","date":"2023-05-23","arxiv_id":"2305.13783","repositories_listed":0,"syntology":null},{"url":null,"slug":"research-on-multi-agent-communication-and","title":"Research on Multi-Agent Communication and Collaborative Decision-Making Based on Deep Reinforcement Learning","date":"2023-05-23","arxiv_id":"2305.17141","repositories_listed":0,"syntology":null},{"url":null,"slug":"rlboost-boosting-supervised-models-using-deep","title":"RLBoost: Boosting Supervised Models using Deep Reinforcement Learning","date":"2023-05-23","arxiv_id":"2305.14115","repositories_listed":0,"syntology":null},{"url":null,"slug":"semantic-aware-transmission-scheduling-a","title":"Semantic-aware Transmission Scheduling: a Monotonicity-driven Deep Reinforcement Learning Approach","date":"2023-05-23","arxiv_id":"2305.13706","repositories_listed":0,"syntology":null},{"url":null,"slug":"dexpbt-scaling-up-dexterous-manipulation-for","title":"DexPBT: Scaling up Dexterous Manipulation for Hand-Arm Systems with Population Based Training","date":"2023-05-20","arxiv_id":"2305.12127","repositories_listed":0,"syntology":null},{"url":null,"slug":"game-theoretical-analysis-of-reviewer-rewards","title":"Game-Theoretical Analysis of Reviewer Rewards in Peer-Review Journal Systems: Analysis and Experimental Evaluation using Deep Reinforcement Learning","date":"2023-05-20","arxiv_id":"2305.12088","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-design-method-of-building-pipeline","title":"Automatic Design Method of Building Pipeline Layout Based on Deep Reinforcement Learning","date":"2023-05-18","arxiv_id":"2305.10760","repositories_listed":0,"syntology":null},{"url":null,"slug":"black-box-targeted-reward-poisoning-attack","title":"Black-Box Targeted Reward Poisoning Attack Against Online Deep Reinforcement Learning","date":"2023-05-18","arxiv_id":"2305.10681","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-metric-tensor-regularized-policy","title":"Deep Metric Tensor Regularized Policy Gradient","date":"2023-05-18","arxiv_id":"2305.11017","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-packgen-a-deep-reinforcement-learning","title":"Deep PackGen: A Deep Reinforcement Learning Framework for Adversarial Network Packet Generation","date":"2023-05-18","arxiv_id":"2305.11039","repositories_listed":0,"syntology":null},{"url":null,"slug":"drl-meets-dsa-networks-convergence-analysis","title":"DRL meets DSA Networks: Convergence Analysis and Its Application to System Design","date":"2023-05-18","arxiv_id":"2305.11237","repositories_listed":0,"syntology":null},{"url":null,"slug":"lyapunov-driven-deep-reinforcement-learning","title":"Lyapunov-Driven Deep Reinforcement Learning for Edge Inference Empowered by Reconfigurable Intelligent Surfaces","date":"2023-05-18","arxiv_id":"2305.10931","repositories_listed":0,"syntology":null},{"url":null,"slug":"collective-large-scale-wind-farm-multivariate","title":"Collective Large-scale Wind Farm Multivariate Power Output Control Based on Hierarchical Communication Multi-Agent Proximal Policy Optimization","date":"2023-05-17","arxiv_id":"2305.10161","repositories_listed":0,"syntology":null},{"url":null,"slug":"curriculum-learning-in-job-shop-scheduling","title":"Curriculum Learning in Job Shop Scheduling using Reinforcement Learning","date":"2023-05-17","arxiv_id":"2305.10192","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-rl-approach-on-task-placement-and","title":"A Deep RL Approach on Task Placement and Scaling of Edge Resources for Cellular Vehicle-to-Network Service Provisioning","date":"2023-05-16","arxiv_id":"2305.09832","repositories_listed":0,"syntology":null},{"url":null,"slug":"boosting-event-extraction-with-denoised","title":"Boosting Event Extraction with Denoised Structure-to-Text Augmentation","date":"2023-05-16","arxiv_id":"2305.09598","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-to-maximize","title":"Deep Reinforcement Learning to Maximize Arterial Usage during Extreme Congestion","date":"2023-05-16","arxiv_id":"2305.09600","repositories_listed":0,"syntology":null},{"url":null,"slug":"identify-estimate-and-bound-the-uncertainty","title":"Identify, Estimate and Bound the Uncertainty of Reinforcement Learning for Autonomous Driving","date":"2023-05-12","arxiv_id":"2305.07487","repositories_listed":0,"syntology":null},{"url":null,"slug":"intelligent-multicast-routing-method-based-on","title":"Intelligent multicast routing method based on multi-agent deep reinforcement learning in SDWN","date":"2023-05-12","arxiv_id":"2305.10440","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-interference","title":"Deep Reinforcement Learning for Interference Management in UAV-based 3D Networks: Potentials and Challenges","date":"2023-05-11","arxiv_id":"2305.07069","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-memory-mapping-using-deep","title":"Optimizing Memory Mapping Using Deep Reinforcement Learning","date":"2023-05-11","arxiv_id":"2305.07440","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-resource-1","title":"Deep Reinforcement Learning Based Resource Allocation for Cloud Native Wireless Network","date":"2023-05-10","arxiv_id":"2305.06249","repositories_listed":0,"syntology":null},{"url":null,"slug":"sequence-agnostic-multi-object-navigation","title":"Sequence-Agnostic Multi-Object Navigation","date":"2023-05-10","arxiv_id":"2305.06178","repositories_listed":0,"syntology":null},{"url":null,"slug":"assessment-of-reinforcement-learning","title":"Assessment of Reinforcement Learning Algorithms for Nuclear Power Plant Fuel Optimization","date":"2023-05-09","arxiv_id":"2305.05812","repositories_listed":0,"syntology":null},{"url":null,"slug":"cooperating-graph-neural-networks-with-deep","title":"Cooperating Graph Neural Networks with Deep Reinforcement Learning for Vaccine Prioritization","date":"2023-05-09","arxiv_id":"2305.05163","repositories_listed":0,"syntology":null},{"url":"/paper/learnable-behavior-control-breaking-atari","slug":"learnable-behavior-control-breaking-atari","title":"Learnable Behavior Control: Breaking Atari Human World Records via Sample-Efficient Behavior Selection","date":"2023-05-09","arxiv_id":"2305.05239","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-deep-rl-for-intraoperative-planning-of","title":"Safe Deep RL for Intraoperative Planning of Pedicle Screw Placement","date":"2023-05-09","arxiv_id":"2305.05354","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-based-reserve","title":"A Deep Reinforcement Learning-based Reserve Optimization in Active Distribution Systems for Tertiary Frequency Regulation","date":"2023-05-07","arxiv_id":"2305.04163","repositories_listed":0,"syntology":null},{"url":null,"slug":"rescue-conversations-from-dead-ends-efficient","title":"Rescue Conversations from Dead-ends: Efficient Exploration for Task-oriented Dialogue Policy Optimization","date":"2023-05-05","arxiv_id":"2305.03262","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-q-learning-based-distribution-network","title":"Deep Q-Learning-based Distribution Network Reconfiguration for Reliability Improvement","date":"2023-05-02","arxiv_id":"2305.01180","repositories_listed":0,"syntology":null},{"url":null,"slug":"bcedge-slo-aware-dnn-inference-services-with","title":"BCEdge: SLO-Aware DNN Inference Services with Adaptive Batching on Edge Platforms","date":"2023-05-01","arxiv_id":"2305.01519","repositories_listed":0,"syntology":null},{"url":null,"slug":"representations-and-exploration-for-deep","title":"Representations and Exploration for Deep Reinforcement Learning using Singular Value Decomposition","date":"2023-05-01","arxiv_id":"2305.00654","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-scheduling-in-iot-driven-smart","title":"Optimal Scheduling in IoT-Driven Smart Isolated Microgrids Based on Deep Reinforcement Learning","date":"2023-04-28","arxiv_id":"2305.00127","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-policy-optimization-in-deep","title":"Adversarial Policy Optimization in Deep Reinforcement Learning","date":"2023-04-27","arxiv_id":"2304.14533","repositories_listed":0,"syntology":null},{"url":null,"slug":"batch-quantum-reinforcement-learning","title":"BCQQ: Batch-Constraint Quantum Q-Learning with Cyclic Data Re-uploading","date":"2023-04-27","arxiv_id":"2305.00905","repositories_listed":0,"syntology":null},{"url":null,"slug":"cooperative-hierarchical-deep-reinforcement","title":"Cooperative Hierarchical Deep Reinforcement Learning based Joint Sleep and Power Control in RIS-aided Energy-Efficient RAN","date":"2023-04-26","arxiv_id":"2304.13226","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-agile-soccer-skills-for-a-bipedal","title":"Learning Agile Soccer Skills for a Bipedal Robot with Deep Reinforcement Learning","date":"2023-04-26","arxiv_id":"2304.13653","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-optimization-framework-for-herbal","title":"A optimization framework for herbal prescription planning based on deep reinforcement learning","date":"2023-04-25","arxiv_id":"2304.12828","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-deep-reinforcement-learning-for-thz","title":"Federated Deep Reinforcement Learning for THz-Beam Search with Limited CSI","date":"2023-04-25","arxiv_id":"2304.13109","repositories_listed":0,"syntology":null},{"url":null,"slug":"roll-drop-accounting-for-observation-noise","title":"Roll-Drop: accounting for observation noise with a single parameter","date":"2023-04-25","arxiv_id":"2304.13150","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-halftoning-via-deep-reinforcement","title":"Efficient Halftoning via Deep Reinforcement Learning","date":"2023-04-24","arxiv_id":"2304.12152","repositories_listed":0,"syntology":null},{"url":null,"slug":"parallel-bootstrap-based-on-policy-deep","title":"Parallel bootstrap-based on-policy deep reinforcement learning for continuous flow control applications","date":"2023-04-24","arxiv_id":"2304.12330","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-declarative-procedural-and","title":"Bridging Declarative, Procedural, and Conditional Metacognitive Knowledge Gap Using Deep Reinforcement Learning","date":"2023-04-23","arxiv_id":"2304.11739","repositories_listed":0,"syntology":null},{"url":"/paper/efficient-deep-reinforcement-learning","slug":"efficient-deep-reinforcement-learning","title":"Efficient Deep Reinforcement Learning Requires Regulating Overfitting","date":"2023-04-20","arxiv_id":"2304.10466","repositories_listed":0,"syntology":{"n":15,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/efficient-deep-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2304.10466","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.10466"}},"official":null}},{"url":null,"slug":"two-memory-reinforcement-learning","title":"Two-Memory Reinforcement Learning","date":"2023-04-20","arxiv_id":"2304.10098","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-agent-for-beyond-visual-range-air","title":"Autonomous Agent for Beyond Visual Range Air Combat: A Deep Reinforcement Learning Approach","date":"2023-04-19","arxiv_id":"2304.09669","repositories_listed":0,"syntology":null},{"url":null,"slug":"long-term-fairness-with-unknown-dynamics","title":"Long-Term Fairness with Unknown Dynamics","date":"2023-04-19","arxiv_id":"2304.09362","repositories_listed":0,"syntology":null},{"url":null,"slug":"alzheimers-disease-diagnosis-using-machine","title":"Alzheimers Disease Diagnosis using Machine Learning: A Review","date":"2023-04-17","arxiv_id":"2304.09178","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-explainable-relational-reinforcement","title":"Deep Explainable Relational Reinforcement Learning: A Neuro-Symbolic Approach","date":"2023-04-17","arxiv_id":"2304.08349","repositories_listed":0,"syntology":null},{"url":null,"slug":"integration-of-reinforcement-learning-based","title":"Integration of Reinforcement Learning Based Behavior Planning With Sampling Based Motion Planning for Automated Driving","date":"2023-04-17","arxiv_id":"2304.08280","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-deep-reinforcement-learning-for","title":"Leveraging Deep Reinforcement Learning for Metacognitive Interventions across Intelligent Tutoring Systems","date":"2023-04-17","arxiv_id":"2304.09821","repositories_listed":0,"syntology":null},{"url":null,"slug":"reclaimer-a-reinforcement-learning-approach","title":"Reclaimer: A Reinforcement Learning Approach to Dynamic Resource Allocation for Cloud Microservices","date":"2023-04-17","arxiv_id":"2304.07941","repositories_listed":0,"syntology":null},{"url":null,"slug":"context-aware-domain-adaptation-for-time","title":"Context-aware Domain Adaptation for Time Series Anomaly Detection","date":"2023-04-15","arxiv_id":"2304.07453","repositories_listed":0,"syntology":null},{"url":null,"slug":"mvco-dot-multi-view-contrastive-domain","title":"MvCo-DoT:Multi-View Contrastive Domain Transfer Network for Medical Report Generation","date":"2023-04-15","arxiv_id":"2304.07465","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-applied-to-an","title":"Deep reinforcement learning applied to an assembly sequence planning problem with user preferences","date":"2023-04-13","arxiv_id":"2304.06567","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-ris-aided-eh-noma-networks-a-deep","title":"Active RIS-aided EH-NOMA Networks: A Deep Reinforcement Learning Approach","date":"2023-04-11","arxiv_id":"2304.12184","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-time-model-free-deep-reinforcement","title":"Real-Time Model-Free Deep Reinforcement Learning for Force Control of a Series Elastic Actuator","date":"2023-04-11","arxiv_id":"2304.04911","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-tutor-better-supported","title":"Reinforcement Learning Tutor Better Supported Lower Performers in a Math Task","date":"2023-04-11","arxiv_id":"2304.04933","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-modular-framework-for-stabilizing-deep","title":"A modular framework for stabilizing deep reinforcement learning control","date":"2023-04-07","arxiv_id":"2304.03422","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-optimal-1","title":"Deep Reinforcement Learning Based Optimal Infinite-Horizon Control of Probabilistic Boolean Control Networks","date":"2023-04-07","arxiv_id":"2304.03489","repositories_listed":0,"syntology":null},{"url":null,"slug":"full-gradient-deep-reinforcement-learning-for","title":"Full Gradient Deep Reinforcement Learning for Average-Reward Criterion","date":"2023-04-07","arxiv_id":"2304.03729","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-vehicle","title":"Deep Reinforcement Learning Based Vehicle Selection for Asynchronous Federated Learning Enabled Vehicular Edge Computing","date":"2023-04-06","arxiv_id":"2304.02832","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-energy-storage-scheduling-for-wind","title":"Optimal Energy Storage Scheduling for Wind Curtailment Reduction and Energy Arbitrage: A Deep Reinforcement Learning Approach","date":"2023-04-05","arxiv_id":"2304.02239","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-irrigation-efficiency-using-deep","title":"Optimizing Irrigation Efficiency using Deep Reinforcement Learning in the Field","date":"2023-04-04","arxiv_id":"2304.01435","repositories_listed":0,"syntology":null},{"url":null,"slug":"regularization-of-the-policy-updates-for","title":"Regularization of the policy updates for stabilizing Mean Field Games","date":"2023-04-04","arxiv_id":"2304.01547","repositories_listed":0,"syntology":null},{"url":null,"slug":"enabling-a-network-ai-gym-for-autonomous","title":"Enabling A Network AI Gym for Autonomous Cyber Agents","date":"2023-04-03","arxiv_id":"2304.01366","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-semantic-aware-sampling-and","title":"Optimal Semantic-aware Sampling and Transmission in Energy Harvesting Systems Through the AoII","date":"2023-04-03","arxiv_id":"2304.00875","repositories_listed":0,"syntology":null},{"url":null,"slug":"unified-emulation-simulation-training","title":"Unified Emulation-Simulation Training Environment for Autonomous Cyber Agents","date":"2023-04-03","arxiv_id":"2304.01244","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-microgrid-collaborative-optimization","title":"Multi-Microgrid Collaborative Optimization Scheduling Using an Improved Multi-Agent Soft Actor-Critic Algorithm","date":"2023-04-01","arxiv_id":"2304.01223","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-complicated-manipulation-skills-via","title":"Learning Complicated Manipulation Skills via Deterministic Policy with Limited Demonstrations","date":"2023-03-29","arxiv_id":"2303.16469","repositories_listed":0,"syntology":null},{"url":null,"slug":"physical-deep-reinforcement-learning-towards","title":"Physical Deep Reinforcement Learning Towards Safety Guarantee","date":"2023-03-29","arxiv_id":"2303.16860","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantum-deep-hedging","title":"Quantum Deep Hedging","date":"2023-03-29","arxiv_id":"2303.16585","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-background-music-for-a-fighting-game","title":"Adaptive Background Music for a Fighting Game: A Multi-Instrument Volume Modulation Approach","date":"2023-03-28","arxiv_id":"2303.15734","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-use-of-reinforcement-learning-for","title":"On the Use of Reinforcement Learning for Attacking and Defending Load Frequency Control","date":"2023-03-28","arxiv_id":"2303.15736","repositories_listed":0,"syntology":null},{"url":null,"slug":"bi-manual-block-assembly-via-sim-to-real","title":"Bi-Manual Block Assembly via Sim-to-Real Reinforcement Learning","date":"2023-03-27","arxiv_id":"2303.14870","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-smoothing-distribution-exploration","title":"Optimal Smoothing Distribution Exploration for Backdoor Neutralization in Deep Learning-based Traffic Systems","date":"2023-03-24","arxiv_id":"2303.14197","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-path-following-on-rivers-using","title":"Robust Path Following on Rivers Using Bootstrapped Reinforcement Learning","date":"2023-03-24","arxiv_id":"2303.15178","repositories_listed":0,"syntology":null},{"url":null,"slug":"connected-superlevel-set-in-deep","title":"Connected Superlevel Set in (Deep) Reinforcement Learning and its Application to Minimax Theorems","date":"2023-03-23","arxiv_id":"2303.12981","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-2","title":"Deep Reinforcement Learning for Localizability-Enhanced Navigation in Dynamic Human Environments","date":"2023-03-22","arxiv_id":"2303.12354","repositories_listed":0,"syntology":null}],"record_sha256":"1c84360e067c347bbc66b322e5114f63bec5551603ff0699442390e48bfcc080","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}