{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/83","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":83,"pages_in_order":152,"rows_per_page":100,"rows":[8201,8300],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/82","next":"/task/reinforcement-learning-1/papers/84","papers":[{"url":null,"slug":"a-contact-safe-reinforcement-learning","title":"A Contact-Safe Reinforcement Learning Framework for Contact-Rich Robot Manipulation","date":"2022-07-27","arxiv_id":"2207.13438","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributional-actor-critic-ensemble-for","title":"Distributional Actor-Critic Ensemble for Uncertainty-Aware Continuous Control","date":"2022-07-27","arxiv_id":"2207.13730","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-shielding-for-reinforcement-learning","title":"Dynamic Shielding for Reinforcement Learning in Black-Box Environments","date":"2022-07-27","arxiv_id":"2207.13446","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-objective-provisioning-of-network","title":"Multi-Objective Provisioning of Network Slices using Deep Reinforcement Learning","date":"2022-07-27","arxiv_id":"2207.13821","repositories_listed":0,"syntology":null},{"url":null,"slug":"poset-rl-phase-ordering-for-optimizing-size","title":"POSET-RL: Phase ordering for Optimizing Size and Execution Time using Reinforcement Learning","date":"2022-07-27","arxiv_id":"2208.04238","repositories_listed":0,"syntology":null},{"url":null,"slug":"structural-similarity-for-improved-transfer","title":"Structural Similarity for Improved Transfer in Reinforcement Learning","date":"2022-07-27","arxiv_id":"2207.13813","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-training-for-neural-tsp-solver","title":"Unsupervised Training for Neural TSP Solver","date":"2022-07-27","arxiv_id":"2207.13667","repositories_listed":0,"syntology":null},{"url":null,"slug":"branch-ranking-for-efficient-mixed-integer","title":"Branch Ranking for Efficient Mixed-Integer Programming via Offline Ranking-based Policy Learning","date":"2022-07-26","arxiv_id":"2207.13701","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-reinforcement-learning-at-multiple","title":"Offline Reinforcement Learning at Multiple Frequencies","date":"2022-07-26","arxiv_id":"2207.13082","repositories_listed":0,"syntology":null},{"url":null,"slug":"planning-and-learning-a-review-of-methods","title":"Planning and Learning: Path-Planning for Autonomous Vehicles, a Review of the Literature","date":"2022-07-26","arxiv_id":"2207.13181","repositories_listed":0,"syntology":null},{"url":null,"slug":"semi-analytical-industrial-cooling-system","title":"Semi-analytical Industrial Cooling System Model for Reinforcement Learning","date":"2022-07-26","arxiv_id":"2207.13131","repositories_listed":0,"syntology":null},{"url":null,"slug":"cooperative-actor-critic-via-td-error","title":"Cooperative Actor-Critic via TD Error Aggregation","date":"2022-07-25","arxiv_id":"2207.12533","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-planning-in-open-ended-dialogue-using","title":"Dynamic Planning in Open-Ended Dialogue using Reinforcement Learning","date":"2022-07-25","arxiv_id":"2208.02294","repositories_listed":0,"syntology":null},{"url":null,"slug":"flowsheet-synthesis-through-hierarchical","title":"Flowsheet synthesis through hierarchical reinforcement learning and graph neural networks","date":"2022-07-25","arxiv_id":"2207.12051","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-neural-ordinary-differential-equations","title":"Adaptive Asynchronous Control Using Meta-learned Neural Ordinary Differential Equations","date":"2022-07-25","arxiv_id":"2207.12062","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-reinforcement-learning-for-periodic","title":"Online Reinforcement Learning for Periodic MDP","date":"2022-07-25","arxiv_id":"2207.12045","repositories_listed":0,"syntology":null},{"url":null,"slug":"repnp-plug-and-play-with-deep-reinforcement","title":"REPNP: Plug-and-Play with Deep Reinforcement Learning Prior for Robust Image Restoration","date":"2022-07-25","arxiv_id":"2207.12056","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-decision-making-at-the-intersection","title":"Adaptive Decision Making at the Intersection for Autonomous Vehicles Based on Skill Discovery","date":"2022-07-24","arxiv_id":"2207.11724","repositories_listed":0,"syntology":null},{"url":null,"slug":"anti-overestimation-dialogue-policy-learning-1","title":"Anti-Overestimation Dialogue Policy Learning for Task-Completion Dialogue System","date":"2022-07-24","arxiv_id":"2207.11762","repositories_listed":0,"syntology":null},{"url":null,"slug":"epersist-a-self-balancing-robot-using-pid","title":"Epersist: A Self Balancing Robot Using PID Controller And Deep Reinforcement Learning","date":"2022-07-23","arxiv_id":"2207.11431","repositories_listed":0,"syntology":null},{"url":null,"slug":"halftoning-with-multi-agent-deep","title":"Halftoning with Multi-Agent Deep Reinforcement Learning","date":"2022-07-23","arxiv_id":"2207.11408","repositories_listed":0,"syntology":null},{"url":null,"slug":"learn-continuously-act-discretely-hybrid","title":"Learn Continuously, Act Discretely: Hybrid Action-Space Reinforcement Learning For Optimal Execution","date":"2022-07-22","arxiv_id":"2207.11152","repositories_listed":0,"syntology":null},{"url":null,"slug":"addressing-optimism-bias-in-sequence-modeling","title":"Addressing Optimism Bias in Sequence Modeling for Reinforcement Learning","date":"2022-07-21","arxiv_id":"2207.10295","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-asset-closed-loop-reservoir-management","title":"Multi-Asset Closed-Loop Reservoir Management Using Deep Reinforcement Learning","date":"2022-07-21","arxiv_id":"2207.10376","repositories_listed":0,"syntology":null},{"url":null,"slug":"strategising-template-guided-needle-placement","title":"Strategising template-guided needle placement for MR-targeted prostate biopsy","date":"2022-07-21","arxiv_id":"2207.10784","repositories_listed":0,"syntology":null},{"url":null,"slug":"subgraph-matching-via-query-conditioned","title":"Detecting Small Query Graphs in A Large Graph via Neural Subgraph Search","date":"2022-07-21","arxiv_id":"2207.10305","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-robust-on-ramp-merging-via-augmented","title":"Towards Robust On-Ramp Merging via Augmented Multimodal Reinforcement Learning","date":"2022-07-21","arxiv_id":"2208.07307","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantifying-the-effect-of-feedback-frequency","title":"Quantifying the Effect of Feedback Frequency in Interactive Reinforcement Learning for Robotic Tasks","date":"2022-07-20","arxiv_id":"2207.09845","repositories_listed":0,"syntology":null},{"url":null,"slug":"actor-critic-based-improper-reinforcement","title":"Actor-Critic based Improper Reinforcement Learning","date":"2022-07-19","arxiv_id":"2207.09090","repositories_listed":0,"syntology":null},{"url":null,"slug":"feasible-adversarial-robust-reinforcement","title":"Feasible Adversarial Robust Reinforcement Learning for Underspecified Environments","date":"2022-07-19","arxiv_id":"2207.09597","repositories_listed":0,"syntology":null},{"url":null,"slug":"few-shot-teamwork","title":"Few-Shot Teamwork","date":"2022-07-19","arxiv_id":"2207.09300","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-action-translator-for-meta","title":"Learning Action Translator for Meta Reinforcement Learning on Sparse-Reward Tasks","date":"2022-07-19","arxiv_id":"2207.09071","repositories_listed":0,"syntology":null},{"url":null,"slug":"new-auction-algorithms-for-path-planning","title":"New Auction Algorithms for Path Planning, Network Transport, and Reinforcement Learning","date":"2022-07-19","arxiv_id":"2207.09588","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-decentralizing-federated-reinforcement","title":"On Decentralizing Federated Reinforcement Learning in Multi-Robot Scenarios","date":"2022-07-19","arxiv_id":"2207.09372","repositories_listed":0,"syntology":null},{"url":null,"slug":"riemannian-stochastic-gradient-method-for","title":"Riemannian Stochastic Gradient Method for Nested Composition Optimization","date":"2022-07-19","arxiv_id":"2207.09350","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-few-expert-queries-suffices-for-sample","title":"A Few Expert Queries Suffices for Sample-Efficient RL with Resets and Linear Value Approximation","date":"2022-07-18","arxiv_id":"2207.08342","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-information-theoretic-analysis-of-bayesian","title":"An Information-Theoretic Analysis of Bayesian Reinforcement Learning","date":"2022-07-18","arxiv_id":"2207.08735","repositories_listed":0,"syntology":null},{"url":null,"slug":"boolean-decision-rules-for-reinforcement","title":"Boolean Decision Rules for Reinforcement Learning Policy Summarisation","date":"2022-07-18","arxiv_id":"2207.08651","repositories_listed":0,"syntology":null},{"url":null,"slug":"mad-for-robust-reinforcement-learning-in-1","title":"MAD for Robust Reinforcement Learning in Machine Translation","date":"2022-07-18","arxiv_id":"2207.08583","repositories_listed":0,"syntology":null},{"url":null,"slug":"mlgoperf-an-ml-guided-inliner-to-optimize","title":"MLGOPerf: An ML Guided Inliner to Optimize Performance","date":"2022-07-18","arxiv_id":"2207.08389","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-stabilizing-reinforcement-learning-without","title":"A framework for online, stabilizing reinforcement learning","date":"2022-07-18","arxiv_id":"2207.08730","repositories_listed":0,"syntology":null},{"url":null,"slug":"context-sequence-theory-a-common-explanation","title":"Context sequence theory: a common explanation for multiple types of learning","date":"2022-07-17","arxiv_id":"2208.04707","repositories_listed":0,"syntology":null},{"url":null,"slug":"dimba-discretely-masked-black-box-attack-in","title":"DIMBA: Discretely Masked Black-Box Attack in Single Object Tracking","date":"2022-07-17","arxiv_id":"2207.08044","repositories_listed":0,"syntology":null},{"url":null,"slug":"minimum-description-length-control","title":"Minimum Description Length Control","date":"2022-07-17","arxiv_id":"2207.08258","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-survival-a","title":"Reinforcement Learning For Survival, A Clinically Motivated Method For Critically Ill Patients","date":"2022-07-17","arxiv_id":"2207.08040","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-action-governor-for-uncertain","title":"Robust Action Governor for Uncertain Piecewise Affine Systems with Non-convex Constraints and Safe Reinforcement Learning","date":"2022-07-17","arxiv_id":"2207.08240","repositories_listed":0,"syntology":null},{"url":null,"slug":"associative-memory-based-experience-replay","title":"Associative Memory Based Experience Replay for Deep Reinforcement Learning","date":"2022-07-16","arxiv_id":"2207.07791","repositories_listed":0,"syntology":null},{"url":null,"slug":"bcrlsp-an-offline-reinforcement-learning","title":"BCRLSP: An Offline Reinforcement Learning Framework for Sequential Targeted Promotion","date":"2022-07-16","arxiv_id":"2207.07790","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-hedging-continuous-reinforcement","title":"Deep Hedging: Continuous Reinforcement Learning for Hedging of General Portfolios across Multiple Risk Aversions","date":"2022-07-15","arxiv_id":"2207.07467","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-data-collection-in-deep","title":"Optimizing Data Collection in Deep Reinforcement Learning","date":"2022-07-15","arxiv_id":"2207.07736","repositories_listed":0,"syntology":null},{"url":null,"slug":"outcome-guided-counterfactuals-for","title":"Outcome-Guided Counterfactuals for Reinforcement Learning Agents from a Jointly Trained Generative Latent Space","date":"2022-07-15","arxiv_id":"2207.07710","repositories_listed":0,"syntology":null},{"url":null,"slug":"skill-based-model-based-reinforcement","title":"Skill-based Model-based Reinforcement Learning","date":"2022-07-15","arxiv_id":"2207.07560","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-nature-of-temporal-difference-errors-in","title":"The Nature of Temporal Difference Errors in Multi-step Distributional Reinforcement Learning","date":"2022-07-15","arxiv_id":"2207.07570","repositories_listed":0,"syntology":null},{"url":null,"slug":"covy-an-ai-powered-robot-for-detection-of","title":"Covy: An AI-powered Robot with a Compound Vision System for Detecting Breaches in Social Distancing","date":"2022-07-14","arxiv_id":"2207.06847","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-deep-reinforcement-learning-2","title":"Multi-Agent Deep Reinforcement Learning-Driven Mitigation of Adverse Effects of Cyber-Attacks on Electric Vehicle Charging Station","date":"2022-07-14","arxiv_id":"2207.07041","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-based-offensive","title":"A Reinforcement Learning-based Offensive semantics Censorship System for Chatbots","date":"2022-07-13","arxiv_id":"2207.10569","repositories_listed":0,"syntology":null},{"url":null,"slug":"continual-meta-reinforcement-learning-for-uav","title":"Continual Meta-Reinforcement Learning for UAV-Aided Vehicular Wireless Networks","date":"2022-07-13","arxiv_id":"2207.06131","repositories_listed":0,"syntology":null},{"url":null,"slug":"griddlyjs-a-web-ide-for-reinforcement","title":"GriddlyJS: A Web IDE for Reinforcement Learning","date":"2022-07-13","arxiv_id":"2207.06105","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-optimization-with-sparse-global","title":"Policy Optimization with Sparse Global Contrastive Explanations","date":"2022-07-13","arxiv_id":"2207.06269","repositories_listed":0,"syntology":null},{"url":null,"slug":"scheduling-out-of-coverage-vehicular","title":"Scheduling Out-of-Coverage Vehicular Communications Using Reinforcement Learning","date":"2022-07-13","arxiv_id":"2207.06537","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-play-psro-toward-optimal-populations-in","title":"Self-Play PSRO: Toward Optimal Populations in Two-Player Zero-Sum Games","date":"2022-07-13","arxiv_id":"2207.06541","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimistic-pac-reinforcement-learning-the","title":"Optimistic PAC Reinforcement Learning: the Instance-Dependent View","date":"2022-07-12","arxiv_id":"2207.05852","repositories_listed":0,"syntology":null},{"url":null,"slug":"pac-reinforcement-learning-for-predictive","title":"PAC Reinforcement Learning for Predictive State Representations","date":"2022-07-12","arxiv_id":"2207.05738","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-global-optimality-in-cooperative-marl","title":"Towards Global Optimality in Cooperative MARL with the Transformation And Distillation Framework","date":"2022-07-12","arxiv_id":"2207.11143","repositories_listed":0,"syntology":null},{"url":null,"slug":"don-t-start-from-scratch-leveraging-prior","title":"Don't Start From Scratch: Leveraging Prior Data to Automate Robotic Reinforcement Learning","date":"2022-07-11","arxiv_id":"2207.04703","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-temporally-extended-skills-in","title":"Learning Temporally Extended Skills in Continuous Domains as Symbolic Actions for Planning","date":"2022-07-11","arxiv_id":"2207.05018","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-unicode-x2013-based","title":"Reinforcement Learning$\\unicode{x2013}$Based Transient Response Shaping for Microgrids","date":"2022-07-11","arxiv_id":"2207.05020","repositories_listed":0,"syntology":null},{"url":null,"slug":"state-dropout-based-curriculum-reinforcement","title":"State Dropout-Based Curriculum Reinforcement Learning for Self-Driving at Unsignalized Intersections","date":"2022-07-10","arxiv_id":"2207.04361","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-long-term-1","title":"Deep Reinforcement Learning for Long-Term Voltage Stability Control","date":"2022-07-09","arxiv_id":"2207.04240","repositories_listed":0,"syntology":null},{"url":null,"slug":"ablation-study-of-how-run-time-assurance","title":"Ablation Study of How Run Time Assurance Impacts the Training and Performance of Reinforcement Learning Agents","date":"2022-07-08","arxiv_id":"2207.04117","repositories_listed":0,"syntology":null},{"url":null,"slug":"high-performance-simulation-for-scalable","title":"High Performance Simulation for Scalable Multi-Agent Reinforcement Learning","date":"2022-07-08","arxiv_id":"2207.03945","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-for-multi-energy","title":"Safe reinforcement learning for multi-energy management systems with known constraint functions","date":"2022-07-08","arxiv_id":"2207.03830","repositories_listed":0,"syntology":null},{"url":null,"slug":"domain-adapting-speech-emotion-recognition","title":"Domain Adapting Deep Reinforcement Learning for Real-world Speech Emotion Recognition","date":"2022-07-07","arxiv_id":"2207.12248","repositories_listed":0,"syntology":null},{"url":null,"slug":"drl-isp-multi-objective-camera-isp-with-deep","title":"DRL-ISP: Multi-Objective Camera ISP with Deep Reinforcement Learning","date":"2022-07-07","arxiv_id":"2207.03081","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-human-like-explanations-for-robot","title":"Evaluating Human-like Explanations for Robot Actions in Reinforcement Learning Scenarios","date":"2022-07-07","arxiv_id":"2207.03214","repositories_listed":0,"syntology":null},{"url":null,"slug":"gym-dssat-a-crop-model-turned-into-a","title":"gym-DSSAT: a crop model turned into a Reinforcement Learning environment","date":"2022-07-07","arxiv_id":"2207.03270","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-objective-optimization-of-notifications","title":"Multi-objective Optimization of Notifications Using Offline Reinforcement Learning","date":"2022-07-07","arxiv_id":"2207.03029","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-distributed","title":"Reinforcement Learning for Distributed Transient Frequency Control with Stability and Safety Guarantees","date":"2022-07-07","arxiv_id":"2207.03329","repositories_listed":0,"syntology":null},{"url":null,"slug":"variational-multiscale-reinforcement-learning","title":"Variational multiscale reinforcement learning for discovering reduced order closure models of nonlinear spatiotemporal transport systems","date":"2022-07-07","arxiv_id":"2207.12854","repositories_listed":0,"syntology":null},{"url":null,"slug":"vessel-following-model-for-inland-waterways","title":"Vessel-following model for inland waterways based on deep reinforcement learning","date":"2022-07-07","arxiv_id":"2207.03257","repositories_listed":0,"syntology":null},{"url":null,"slug":"inferring-and-conveying-intentionality-beyond","title":"Inferring and Conveying Intentionality: Beyond Numerical Rewards to Logical Intentions","date":"2022-07-06","arxiv_id":"2207.05058","repositories_listed":0,"syntology":null},{"url":null,"slug":"instance-dependent-near-optimal-policy","title":"Instance-Dependent Near-Optimal Policy Identification in Linear MDPs via Online Experiment Design","date":"2022-07-06","arxiv_id":"2207.02575","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-selection-in-reinforcement-learning","title":"Model Selection in Reinforcement Learning with General Function Approximations","date":"2022-07-06","arxiv_id":"2207.02992","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-portfolio-manager","title":"Reinforcement Learning Portfolio Manager Framework with Monte Carlo Simulation","date":"2022-07-06","arxiv_id":"2207.02458","repositories_listed":0,"syntology":null},{"url":null,"slug":"avddpg-federated-reinforcement-learning","title":"AVDDPG: Federated reinforcement learning applied to autonomous platoon control","date":"2022-07-05","arxiv_id":"2207.03484","repositories_listed":0,"syntology":null},{"url":null,"slug":"decentralized-scheduling-through-an-adaptive","title":"Decentralized scheduling through an adaptive, trading-based multi-agent system","date":"2022-07-05","arxiv_id":"2207.11172","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-approach-for","title":"Deep Reinforcement Learning Approach for Trading Automation in The Stock Market","date":"2022-07-05","arxiv_id":"2208.07165","repositories_listed":0,"syntology":null},{"url":null,"slug":"explainability-in-deep-reinforcement-learning-1","title":"Explainability in Deep Reinforcement Learning, a Review into Current Methods and Applications","date":"2022-07-05","arxiv_id":"2207.01911","repositories_listed":0,"syntology":null},{"url":null,"slug":"planning-with-rl-and-episodic-memory","title":"Planning with RL and episodic-memory behavioral priors","date":"2022-07-05","arxiv_id":"2207.01845","repositories_listed":0,"syntology":null},{"url":null,"slug":"resource-allocation-in-multicore-elastic","title":"Resource Allocation in Multicore Elastic Optical Networks: A Deep Reinforcement Learning Approach","date":"2022-07-05","arxiv_id":"2207.02074","repositories_listed":0,"syntology":null},{"url":null,"slug":"tackling-real-world-autonomous-driving-using","title":"Tackling Real-World Autonomous Driving using Deep Reinforcement Learning","date":"2022-07-05","arxiv_id":"2207.02162","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-via-confidence","title":"Safe Reinforcement Learning via Confidence-Based Filters","date":"2022-07-04","arxiv_id":"2207.01337","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-bellman-hedging","title":"Deep Bellman Hedging","date":"2022-07-03","arxiv_id":"2207.00932","repositories_listed":0,"syntology":null},{"url":null,"slug":"government-intervention-in-catastrophe","title":"Government Intervention in Catastrophe Insurance Markets: A Reinforcement Learning Approach","date":"2022-07-03","arxiv_id":"2207.01010","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-learning-in-continuous-time","title":"q-Learning in Continuous Time","date":"2022-07-02","arxiv_id":"2207.00713","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-approaches-for-the","title":"Reinforcement Learning Approaches for the Orienteering Problem with Stochastic and Dynamic Release Dates","date":"2022-07-02","arxiv_id":"2207.00885","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-for-a-robot-being","title":"Safe Reinforcement Learning for a Robot Being Pursued but with Objectives Covering More Than Capture-avoidance","date":"2022-07-02","arxiv_id":"2207.00842","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-shoulder-to-cry-on-towards-a-motivational","title":"A Shoulder to Cry on: Towards A Motivational Virtual Assistant for Assuaging Mental Agony","date":"2022-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"action-modulated-midbrain-dopamine-activity","title":"Action-modulated midbrain dopamine activity arises from distributed control policies","date":"2022-07-01","arxiv_id":"2207.00636","repositories_listed":0,"syntology":null},{"url":null,"slug":"agent-with-tangent-based-formulation-and","title":"Agent with Tangent-based Formulation and Anatomical Perception for Standard Plane Localization in 3D Ultrasound","date":"2022-07-01","arxiv_id":"2207.00475","repositories_listed":0,"syntology":null}],"record_sha256":"8bc43d9a267bb7fdc3128d04936dbd3d00af379a284ea08540e4978119e2cedf","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}