{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/81","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":81,"pages_in_order":152,"rows_per_page":100,"rows":[8001,8100],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/80","next":"/task/reinforcement-learning-1/papers/82","papers":[{"url":null,"slug":"an-information-theoretic-perspective-on-3","title":"An information-theoretic perspective on intrinsic motivation in reinforcement learning: a survey","date":"2022-09-19","arxiv_id":"2209.08890","repositories_listed":0,"syntology":null},{"url":null,"slug":"enforcing-the-consensus-between-trajectory","title":"Enforcing the consensus between Trajectory Optimization and Policy Learning for precise robot control","date":"2022-09-19","arxiv_id":"2209.09006","repositories_listed":0,"syntology":null},{"url":null,"slug":"guess-what-i-m-doing-extending-legibility-to","title":"\"Guess what I'm doing\": Extending legibility to sequential decision tasks","date":"2022-09-19","arxiv_id":"2209.09141","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-for-adaptive-3","title":"Meta-Reinforcement Learning for Adaptive Control of Second Order Systems","date":"2022-09-19","arxiv_id":"2209.09301","repositories_listed":0,"syntology":null},{"url":null,"slug":"msviper-improved-policy-distillation-for","title":"MSVIPER: Improved Policy Distillation for Reinforcement-Learning-Based Robot Navigation","date":"2022-09-19","arxiv_id":"2209.09079","repositories_listed":0,"syntology":null},{"url":null,"slug":"rewarding-episodic-visitation-discrepancy-for","title":"Rewarding Episodic Visitation Discrepancy for Exploration in Reinforcement Learning","date":"2022-09-19","arxiv_id":"2209.08842","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-control-for","title":"Safe reinforcement learning control for continuous-time nonlinear systems without a backup controller","date":"2022-09-19","arxiv_id":"2209.08922","repositories_listed":0,"syntology":null},{"url":null,"slug":"transferring-knowledge-for-reinforcement","title":"Transferring Knowledge for Reinforcement Learning in Contact-Rich Manipulation","date":"2022-09-19","arxiv_id":"2210.02891","repositories_listed":0,"syntology":null},{"url":null,"slug":"evolutionary-deep-reinforcement-learning","title":"Evolutionary Deep Reinforcement Learning Using Elite Buffer: A Novel Approach Towards DRL Combined with EA in Continuous Control Tasks","date":"2022-09-18","arxiv_id":"2209.08480","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-level-explanation-of-deep-reinforcement","title":"Multi-level Explanation of Deep Reinforcement Learning-based Scheduling","date":"2022-09-18","arxiv_id":"2209.09645","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-reinforcement-learning-with-4","title":"Offline Reinforcement Learning with Instrumental Variables in Confounded Markov Decision Processes","date":"2022-09-18","arxiv_id":"2209.08666","repositories_listed":0,"syntology":null},{"url":null,"slug":"simplifying-model-based-rl-learning","title":"Simplifying Model-based RL: Learning Representations, Latent-space Models, and Policies with One Objective","date":"2022-09-18","arxiv_id":"2209.08466","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-robust-and-constrained-multi-agent","title":"A Robust and Constrained Multi-Agent Reinforcement Learning Electric Vehicle Rebalancing Method in AMoD Systems","date":"2022-09-17","arxiv_id":"2209.08230","repositories_listed":0,"syntology":null},{"url":null,"slug":"intrinsically-motivated-reinforcement-2","title":"Intrinsically Motivated Reinforcement Learning based Recommendation with Counterfactual Data Augmentation","date":"2022-09-17","arxiv_id":"2209.08228","repositories_listed":0,"syntology":null},{"url":null,"slug":"ma2ql-a-minimalist-approach-to-fully","title":"MA2QL: A Minimalist Approach to Fully Decentralized Multi-Agent Reinforcement Learning","date":"2022-09-17","arxiv_id":"2209.08244","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-efficient-multi-agent-reinforcement","title":"Sample-Efficient Multi-Agent Reinforcement Learning with Demonstrations for Flocking Control","date":"2022-09-17","arxiv_id":"2209.08351","repositories_listed":0,"syntology":null},{"url":null,"slug":"sub-optimal-policy-aided-multi-agent","title":"Sub-optimal Policy Aided Multi-Agent Reinforcement Learning for Flocking Control","date":"2022-09-17","arxiv_id":"2209.08347","repositories_listed":0,"syntology":null},{"url":null,"slug":"conservative-dual-policy-optimization-for","title":"Conservative Dual Policy Optimization for Efficient Model-Based Reinforcement Learning","date":"2022-09-16","arxiv_id":"2209.07676","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-inversion-attacks-against-graph-neural","title":"Model Inversion Attacks against Graph Neural Networks","date":"2022-09-16","arxiv_id":"2209.07807","repositories_listed":0,"syntology":null},{"url":null,"slug":"neuromuscular-reinforcement-learning-to","title":"Neuromuscular Reinforcement Learning to Actuate Human Limbs through FES","date":"2022-09-16","arxiv_id":"2209.07849","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-industrial-hvac-systems-with","title":"Optimizing Industrial HVAC Systems with Hierarchical Reinforcement Learning","date":"2022-09-16","arxiv_id":"2209.08112","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-cooperative-p2p","title":"Reinforcement Learning-Based Cooperative P2P Power Trading between DC Nanogrid Clusters with Wind and PV Energy Resources","date":"2022-09-16","arxiv_id":"2209.07744","repositories_listed":0,"syntology":null},{"url":null,"slug":"trustworthy-reinforcement-learning-against","title":"Trustworthy Reinforcement Learning Against Intrinsic Vulnerabilities: Robustness, Safety, and Generalizability","date":"2022-09-16","arxiv_id":"2209.08025","repositories_listed":0,"syntology":null},{"url":null,"slug":"value-summation-a-novel-scoring-function-for","title":"Value Summation: A Novel Scoring Function for MPC-based Model-based Reinforcement Learning","date":"2022-09-16","arxiv_id":"2209.08169","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-offline-reinforcement-learning-help","title":"Can Offline Reinforcement Learning Help Natural Language Understanding?","date":"2022-09-15","arxiv_id":"2212.03864","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-task-1","title":"Deep Reinforcement Learning for Task Offloading in UAV-Aided Smart Farm Networks","date":"2022-09-15","arxiv_id":"2209.07367","repositories_listed":0,"syntology":null},{"url":null,"slug":"iot-aerial-base-station-task-offloading-with","title":"IoT-Aerial Base Station Task Offloading with Risk-Sensitive Reinforcement Learning for Smart Agriculture","date":"2022-09-15","arxiv_id":"2209.07382","repositories_listed":0,"syntology":null},{"url":null,"slug":"mean-field-approximation-of-cooperative","title":"Mean-Field Approximation of Cooperative Constrained Multi-Agent Reinforcement Learning (CMARL)","date":"2022-09-15","arxiv_id":"2209.07437","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixrts-toward-interpretable-multi-agent","title":"MIXRTs: Toward Interpretable Multi-Agent Reinforcement Learning via Mixing Recurrent Soft Decision Trees","date":"2022-09-15","arxiv_id":"2209.07225","repositories_listed":0,"syntology":null},{"url":null,"slug":"proapt-projection-of-apt-threats-with-deep","title":"ProAPT: Projection of APT Threats with Deep Reinforcement Learning","date":"2022-09-15","arxiv_id":"2209.07215","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-task-driven-robotic-swarm-control","title":"Scalable Task-Driven Robotic Swarm Control via Collision Avoidance and Learning Mean-Field Control","date":"2022-09-15","arxiv_id":"2209.07420","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-deep-neural-function","title":"Understanding Deep Neural Function Approximation in Reinforcement Learning via $ε$-Greedy Exploration","date":"2022-09-15","arxiv_id":"2209.07376","repositories_listed":0,"syntology":null},{"url":null,"slug":"analysis-of-reinforcement-learning-for","title":"Analysis of Reinforcement Learning for determining task replication in workflows","date":"2022-09-14","arxiv_id":"2209.13531","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributionally-robust-offline-reinforcement","title":"Distributionally Robust Offline Reinforcement Learning with Linear Function Approximation","date":"2022-09-14","arxiv_id":"2209.06620","repositories_listed":0,"syntology":null},{"url":null,"slug":"feature-rich-long-term-bitcoin-trading","title":"Feature-Rich Long-term Bitcoin Trading Assistant","date":"2022-09-14","arxiv_id":"2209.12664","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-constrained-reinforcement-learning","title":"Robust Constrained Reinforcement Learning","date":"2022-09-14","arxiv_id":"2209.06866","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-new-reinforcement-learning-framework-to","title":"A new Reinforcement Learning framework to discover natural flavor molecules","date":"2022-09-13","arxiv_id":"2209.05859","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-perception-applied-to-unmanned-aerial","title":"Active Perception Applied To Unmanned Aerial Vehicles Through Deep Reinforcement Learning","date":"2022-09-13","arxiv_id":"2209.06336","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-efficient-reinforcement-learning-and","title":"Data efficient reinforcement learning and adaptive optimal perimeter control of network traffic dynamics","date":"2022-09-13","arxiv_id":"2209.05726","repositories_listed":0,"syntology":null},{"url":null,"slug":"designing-biological-sequences-via-meta","title":"Designing Biological Sequences via Meta-Reinforcement Learning and Bayesian Optimization","date":"2022-09-13","arxiv_id":"2209.06259","repositories_listed":0,"syntology":null},{"url":null,"slug":"skip-training-for-multi-agent-reinforcement","title":"Skip Training for Multi-Agent Reinforcement Learning Controller for Industrial Wave Energy Converters","date":"2022-09-13","arxiv_id":"2209.05656","repositories_listed":0,"syntology":null},{"url":null,"slug":"unifying-causal-inference-and-reinforcement","title":"Unifying Causal Inference and Reinforcement Learning using Higher-Order Category Theory","date":"2022-09-13","arxiv_id":"2209.06262","repositories_listed":0,"syntology":null},{"url":null,"slug":"checklist-models-for-improved-output-fluency","title":"Checklist Models for Improved Output Fluency in Piano Fingering Prediction","date":"2022-09-12","arxiv_id":"2209.05622","repositories_listed":0,"syntology":null},{"url":null,"slug":"deterministic-sequencing-of-exploration-and","title":"Deterministic Sequencing of Exploration and Exploitation for Reinforcement Learning","date":"2022-09-12","arxiv_id":"2209.05408","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-sequential-information","title":"Self-supervised Sequential Information Bottleneck for Robust Exploration in Deep Reinforcement Learning","date":"2022-09-12","arxiv_id":"2209.05333","repositories_listed":0,"syntology":null},{"url":null,"slug":"pathfinding-in-random-partially-observable","title":"Pathfinding in Random Partially Observable Environments with Vision-Informed Deep Reinforcement Learning","date":"2022-09-11","arxiv_id":"2209.04801","repositories_listed":0,"syntology":null},{"url":null,"slug":"performance-driven-controller-tuning-via","title":"Performance-Driven Controller Tuning via Derivative-Free Reinforcement Learning","date":"2022-09-11","arxiv_id":"2209.04854","repositories_listed":0,"syntology":null},{"url":null,"slug":"cooperation-and-competition-flocking-with","title":"Cooperation and Competition: Flocking with Evolutionary Multi-Agent Reinforcement Learning","date":"2022-09-10","arxiv_id":"2209.04696","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-with-contrastive","title":"Safe Reinforcement Learning with Contrastive Risk Prediction","date":"2022-09-10","arxiv_id":"2209.09648","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-memory-related-multi-task-method-based-on","title":"Task-Agnostic Learning to Accomplish New Tasks","date":"2022-09-09","arxiv_id":"2209.04100","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-analysis-of-deep-reinforcement-learning","title":"An Analysis of Deep Reinforcement Learning Agents for Text-based Games","date":"2022-09-09","arxiv_id":"2209.04105","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixed-mathcal-h-2-mathcal-h-infty-lq-games","title":"Robust Policy Optimization in Continuous-time Mixed $\\mathcal{H}_2/\\mathcal{H}_\\infty$ Stochastic Control","date":"2022-09-09","arxiv_id":"2209.04477","repositories_listed":0,"syntology":null},{"url":null,"slug":"rasr-risk-averse-soft-robust-mdps-with-evar","title":"RASR: Risk-Averse Soft-Robust MDPs with EVaR and Entropic Risk","date":"2022-09-09","arxiv_id":"2209.04067","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-large-population-systems-and","title":"A Survey on Large-Population Systems and Scalable Multi-Agent Reinforcement Learning","date":"2022-09-08","arxiv_id":"2209.03859","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybrid-supervised-and-reinforcement-learning","title":"Hybrid Supervised and Reinforcement Learning for the Design and Optimization of Nanophotonic Structures","date":"2022-09-08","arxiv_id":"2209.04447","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-mesh-generation-for-a-blade-passage","title":"Non-iterative generation of an optimal mesh for a blade passage using deep reinforcement learning","date":"2022-09-08","arxiv_id":"2209.05280","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-strategy-for","title":"A Deep Reinforcement Learning Strategy for UAV Autonomous Landing on a Platform","date":"2022-09-07","arxiv_id":"2209.02954","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-sumo-framework-for-deep-reinforcement","title":"A SUMO Framework for Deep Reinforcement Learning Experiments Solving Electric Vehicle Charging Dispatching Problem","date":"2022-09-07","arxiv_id":"2209.02921","repositories_listed":0,"syntology":null},{"url":null,"slug":"concept-modulated-model-based-offline","title":"Concept-modulated model-based offline reinforcement learning for rapid generalization","date":"2022-09-07","arxiv_id":"2209.03207","repositories_listed":0,"syntology":null},{"url":null,"slug":"dc-mrta-decentralized-multi-robot-task","title":"DC-MRTA: Decentralized Multi-Robot Task Allocation and Navigation in Complex Environments","date":"2022-09-07","arxiv_id":"2209.02865","repositories_listed":0,"syntology":null},{"url":null,"slug":"distilling-deep-rl-models-into-interpretable","title":"Distilling Deep RL Models Into Interpretable Neuro-Fuzzy Systems","date":"2022-09-07","arxiv_id":"2209.03357","repositories_listed":0,"syntology":null},{"url":null,"slug":"energy-optimization-of-wind-turbines-via-a","title":"Energy Optimization of Wind Turbines via a Neural Control Policy Based on Reinforcement Learning Markov Chain Monte Carlo Algorithm","date":"2022-09-07","arxiv_id":"2209.03485","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-near-optimality-of-local-policies-in","title":"On the Near-Optimality of Local Policies in Large Cooperative Multi-Agent Reinforcement Learning","date":"2022-09-07","arxiv_id":"2209.03491","repositories_listed":0,"syntology":null},{"url":null,"slug":"finite-time-error-bounds-for-greedy-gq","title":"Finite-Time Error Bounds for Greedy-GQ","date":"2022-09-06","arxiv_id":"2209.02555","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-assistive-robotics-with-deep","title":"Improving Assistive Robotics with Deep Reinforcement Learning","date":"2022-09-05","arxiv_id":"2209.02160","repositories_listed":0,"syntology":null},{"url":null,"slug":"natural-policy-gradients-in-reinforcement","title":"Natural Policy Gradients In Reinforcement Learning Explained","date":"2022-09-05","arxiv_id":"2209.01820","repositories_listed":0,"syntology":null},{"url":null,"slug":"prediction-based-decision-making-for","title":"Prediction Based Decision Making for Autonomous Highway Driving","date":"2022-09-05","arxiv_id":"2209.02106","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-optimised","title":"Reinforcement learning-based optimised control for tracking of nonlinear systems with adversarial attacks","date":"2022-09-05","arxiv_id":"2209.02165","repositories_listed":0,"syntology":null},{"url":null,"slug":"slatefree-a-model-free-decomposition-for","title":"SlateFree: a Model-Free Decomposition for Reinforcement Learning with Slate Actions","date":"2022-09-05","arxiv_id":"2209.01876","repositories_listed":0,"syntology":null},{"url":null,"slug":"variational-inference-for-model-free-and","title":"Variational Inference for Model-Free and Model-Based Reinforcement Learning","date":"2022-09-04","arxiv_id":"2209.01693","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-free-deep-reinforcement-learning-in","title":"Model-Free Deep Reinforcement Learning in Software-Defined Networks","date":"2022-09-03","arxiv_id":"2209.01490","repositories_listed":0,"syntology":null},{"url":null,"slug":"statistical-csi-based-beamforming-for-ris","title":"Statistical CSI-based Beamforming for RIS-Aided Multiuser MISO Systems using Deep Reinforcement Learning","date":"2022-09-03","arxiv_id":"2209.09856","repositories_listed":0,"syntology":null},{"url":null,"slug":"dialogue-evaluation-with-offline","title":"Dialogue Evaluation with Offline Reinforcement Learning","date":"2022-09-02","arxiv_id":"2209.00876","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-practical-communication-strategies","title":"Learning Practical Communication Strategies in Cooperative Multi-Agent Reinforcement Learning","date":"2022-09-02","arxiv_id":"2209.01288","repositories_listed":0,"syntology":null},{"url":null,"slug":"semi-centralised-multi-agent-reinforcement","title":"Taming Multi-Agent Reinforcement Learning with Estimator Variance Reduction","date":"2022-09-02","arxiv_id":"2209.01054","repositories_listed":0,"syntology":null},{"url":null,"slug":"targf-learning-target-gradient-field-for","title":"TarGF: Learning Target Gradient Field to Rearrange Objects without Explicit Goal Specification","date":"2022-09-02","arxiv_id":"2209.00853","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-technique-to-create-weaker-abstract-board","title":"A Technique to Create Weaker Abstract Board Game Agents via Reinforcement Learning","date":"2022-09-01","arxiv_id":"2209.00711","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-quantum-1","title":"Deep reinforcement learning for quantum multiparameter estimation","date":"2022-09-01","arxiv_id":"2209.00671","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamics-adaptive-continual-reinforcement","title":"Dynamics-Adaptive Continual Reinforcement Learning via Progressive Contextualization","date":"2022-09-01","arxiv_id":"2209.00347","repositories_listed":0,"syntology":null},{"url":null,"slug":"metatrader-an-reinforcement-learning-approach","title":"MetaTrader: An Reinforcement Learning Approach Integrating Diverse Policies for Portfolio Optimization","date":"2022-09-01","arxiv_id":"2210.01774","repositories_listed":0,"syntology":null},{"url":null,"slug":"systems-theoretic-process-analysis-of-a-run","title":"Systems Theoretic Process Analysis of a Run Time Assured Neural Network Control System","date":"2022-09-01","arxiv_id":"2209.00552","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-stabilizing-reinforcement-learning-approach","title":"A stabilizing reinforcement learning approach for sampled systems with partially unknown models","date":"2022-08-31","arxiv_id":"2208.14714","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-anomaly-detection-and-search-via","title":"Deep Anomaly Detection and Search via Reinforcement Learning","date":"2022-08-31","arxiv_id":"2208.14834","repositories_listed":0,"syntology":null},{"url":null,"slug":"transmit-power-control-for-indoor-small-cells","title":"Transmit Power Control for Indoor Small Cells: A Method Based on Federated Reinforcement Learning","date":"2022-08-31","arxiv_id":"2209.13536","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-further-exploration-of-deep-multi-agent","title":"A further exploration of deep Multi-Agent Reinforcement Learning with Hybrid Action Space","date":"2022-08-30","arxiv_id":"2208.14447","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-analysis-of-abstracted-model-based-1","title":"An Analysis of Model-Based Reinforcement Learning From Abstracted Observations","date":"2022-08-30","arxiv_id":"2208.14407","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-supervised-continual-learning-a-review","title":"Beyond Supervised Continual Learning: a Review","date":"2022-08-30","arxiv_id":"2208.14307","repositories_listed":0,"syntology":null},{"url":null,"slug":"digital-twin-assisted-risk-aware-sleep-mode","title":"Digital Twin Assisted Risk-Aware Sleep Mode Management Using Deep Q-Networks","date":"2022-08-30","arxiv_id":"2208.14380","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-ensembles-of-reinforcement","title":"Distributed Ensembles of Reinforcement Learning Agents for Electricity Control","date":"2022-08-30","arxiv_id":"2208.14338","repositories_listed":0,"syntology":null},{"url":null,"slug":"evolutionary-deep-reinforcement-learning-for","title":"Evolutionary Deep Reinforcement Learning for Dynamic Slice Management in O-RAN","date":"2022-08-30","arxiv_id":"2208.14394","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-reinforcement-learning-with-sindy","title":"Model-Based Reinforcement Learning with SINDy","date":"2022-08-30","arxiv_id":"2208.14501","repositories_listed":0,"syntology":null},{"url":null,"slug":"categorical-semantics-of-compositional","title":"Categorical semantics of compositional reinforcement learning","date":"2022-08-29","arxiv_id":"2208.13687","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-hardware-security","title":"Reinforcement Learning for Hardware Security: Opportunities, Developments, and Challenges","date":"2022-08-29","arxiv_id":"2208.13885","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-the-limits-of-poisoning-attacks","title":"Understanding the Limits of Poisoning Attacks in Episodic Reinforcement Learning","date":"2022-08-29","arxiv_id":"2208.13663","repositories_listed":0,"syntology":null},{"url":null,"slug":"normality-guided-distributional-reinforcement","title":"Normality-Guided Distributional Reinforcement Learning for Continuous Control","date":"2022-08-28","arxiv_id":"2208.13125","repositories_listed":0,"syntology":null},{"url":null,"slug":"rl-distprivacy-privacy-aware-distributed-deep","title":"RL-DistPrivacy: Privacy-Aware Distributed Deep Inference for low latency IoT systems","date":"2022-08-27","arxiv_id":"2208.13032","repositories_listed":0,"syntology":null},{"url":null,"slug":"supervisorbot-nlp-annotated-real-time","title":"SupervisorBot: NLP-Annotated Real-Time Recommendations of Psychotherapy Treatment Strategies with Deep Reinforcement Learning","date":"2022-08-27","arxiv_id":"2208.13077","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-approach-to-implement-reinforcement","title":"An approach to implement Reinforcement Learning for Heterogeneous Vehicular Networks","date":"2022-08-26","arxiv_id":"2208.12466","repositories_listed":0,"syntology":null},{"url":null,"slug":"attrition-attacking-static-hardware-trojan","title":"ATTRITION: Attacking Static Hardware Trojan Detection Techniques Using Reinforcement Learning","date":"2022-08-26","arxiv_id":"2208.12897","repositories_listed":0,"syntology":null},{"url":null,"slug":"ch-marl-a-multimodal-benchmark-for","title":"CH-MARL: A Multimodal Benchmark for Cooperative, Heterogeneous Multi-Agent Reinforcement Learning","date":"2022-08-26","arxiv_id":"2208.13626","repositories_listed":0,"syntology":null}],"record_sha256":"050545afac26722a52df423bc921945144d5658dc5c8a8f7dcfa9d1e45e776b7","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}