{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/76","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":76,"pages_in_order":135,"rows_per_page":100,"rows":[7501,7600],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/75","next":"/task/reinforcement-learning-2/papers/77","papers":[{"url":null,"slug":"paused-agent-replay-refresh","title":"Paused Agent Replay Refresh","date":"2022-09-26","arxiv_id":"2209.13398","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-adaptive-mesh","title":"Deep Reinforcement Learning for Adaptive Mesh Refinement","date":"2022-09-25","arxiv_id":"2209.12351","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-opportunities-and-challenges-of-using","title":"Opportunities and Challenges from Using Animal Videos in Reinforcement Learning for Navigation","date":"2022-09-25","arxiv_id":"2209.12347","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-learning-using-structural-motifs-in","title":"Reward Learning using Structural Motifs in Inverse Reinforcement Learning","date":"2022-09-25","arxiv_id":"2209.13489","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-lifelong-adaptive-inverse-reinforcement","title":"Fast Lifelong Adaptive Inverse Reinforcement Learning from Demonstrations","date":"2022-09-24","arxiv_id":"2209.11908","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantification-before-selection-active","title":"Quantification before Selection: Active Dynamics Preference for Robust Reinforcement Learning","date":"2022-09-23","arxiv_id":"2209.11596","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-real-world-reinforcement-learning-for","title":"SAFER: Safe Collision Avoidance using Focused and Efficient Trajectory Search with Reinforcement Learning","date":"2022-09-23","arxiv_id":"2209.11789","repositories_listed":0,"syntology":null},{"url":null,"slug":"developing-evaluating-and-scaling-learning","title":"Developing, Evaluating and Scaling Learning Agents in Multi-Agent Environments","date":"2022-09-22","arxiv_id":"2209.10958","repositories_listed":0,"syntology":null},{"url":null,"slug":"minimizing-human-assistance-augmenting-a","title":"Minimizing Human Assistance: Augmenting a Single Demonstration for Deep Reinforcement Learning","date":"2022-09-22","arxiv_id":"2209.11275","repositories_listed":0,"syntology":null},{"url":null,"slug":"parallel-reinforcement-learning-simulation","title":"Parallel Reinforcement Learning Simulation for Visual Quadrotor Navigation","date":"2022-09-22","arxiv_id":"2209.11094","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-computing-and","title":"Reinforcement Learning in Computing and Network Convergence Orchestration","date":"2022-09-22","arxiv_id":"2209.10753","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-framework-with","title":"ECSAS: Exploring Critical Scenarios from Action Sequence in Autonomous Driving","date":"2022-09-21","arxiv_id":"2209.10078","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluation-of-look-ahead-economic-dispatch","title":"Evaluation of Look-ahead Economic Dispatch Using Reinforcement Learning","date":"2022-09-21","arxiv_id":"2209.10207","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-decision-transformer","title":"Hierarchical Decision Transformer","date":"2022-09-21","arxiv_id":"2209.10447","repositories_listed":0,"syntology":null},{"url":null,"slug":"lamarckian-platform-pushing-the-boundaries-of","title":"Lamarckian Platform: Pushing the Boundaries of Evolutionary Reinforcement Learning towards Asynchronous Commercial Games","date":"2022-09-21","arxiv_id":"2209.10055","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-from-symmetry-meta-reinforcement","title":"Learning from Symmetry: Meta-Reinforcement Learning with Symmetrical Behaviors and Language Instructions","date":"2022-09-21","arxiv_id":"2209.10656","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-free-reinforcement-learning-for-asset","title":"Model-Free Reinforcement Learning for Asset Allocation","date":"2022-09-21","arxiv_id":"2209.10458","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-convergence-theory-of-meta","title":"On the Convergence Theory of Meta Reinforcement Learning with Personalized Policies","date":"2022-09-21","arxiv_id":"2209.10072","repositories_listed":0,"syntology":null},{"url":null,"slug":"asynchronous-actor-critic-for-multi-agent","title":"Asynchronous Actor-Critic for Multi-Agent Reinforcement Learning","date":"2022-09-20","arxiv_id":"2209.10113","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-q-network-for-ai-soccer","title":"Deep Q-Network for AI Soccer","date":"2022-09-20","arxiv_id":"2209.09491","repositories_listed":0,"syntology":null},{"url":null,"slug":"locally-constrained-representations-in","title":"Locally Constrained Representations in Reinforcement Learning","date":"2022-09-20","arxiv_id":"2209.09441","repositories_listed":0,"syntology":null},{"url":null,"slug":"macro-action-based-multi-agent-robot-deep","title":"Macro-Action-Based Multi-Agent/Robot Deep Reinforcement Learning under Partial Observability","date":"2022-09-20","arxiv_id":"2209.10003","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-crop-management-with-reinforcement","title":"Optimizing Crop Management with Reinforcement Learning and Imitation Learning","date":"2022-09-20","arxiv_id":"2209.09991","repositories_listed":0,"syntology":null},{"url":null,"slug":"soft-action-priors-towards-robust-policy","title":"Soft Action Priors: Towards Robust Policy Transfer","date":"2022-09-20","arxiv_id":"2209.09882","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-task-prioritized-policy-composition","title":"Towards Task-Prioritized Policy Composition","date":"2022-09-20","arxiv_id":"2209.09536","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-predicting-coding-brain-inspired","title":"Active Predicting Coding: Brain-Inspired Reinforcement Learning for Sparse Reward Robotic Control Problems","date":"2022-09-19","arxiv_id":"2209.09174","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-information-theoretic-perspective-on-3","title":"An information-theoretic perspective on intrinsic motivation in reinforcement learning: a survey","date":"2022-09-19","arxiv_id":"2209.08890","repositories_listed":0,"syntology":null},{"url":null,"slug":"guess-what-i-m-doing-extending-legibility-to","title":"\"Guess what I'm doing\": Extending legibility to sequential decision tasks","date":"2022-09-19","arxiv_id":"2209.09141","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-for-adaptive-3","title":"Meta-Reinforcement Learning for Adaptive Control of Second Order Systems","date":"2022-09-19","arxiv_id":"2209.09301","repositories_listed":0,"syntology":null},{"url":null,"slug":"msviper-improved-policy-distillation-for","title":"MSVIPER: Improved Policy Distillation for Reinforcement-Learning-Based Robot Navigation","date":"2022-09-19","arxiv_id":"2209.09079","repositories_listed":0,"syntology":null},{"url":null,"slug":"rewarding-episodic-visitation-discrepancy-for","title":"Rewarding Episodic Visitation Discrepancy for Exploration in Reinforcement Learning","date":"2022-09-19","arxiv_id":"2209.08842","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-control-for","title":"Safe reinforcement learning control for continuous-time nonlinear systems without a backup controller","date":"2022-09-19","arxiv_id":"2209.08922","repositories_listed":0,"syntology":null},{"url":null,"slug":"transferring-knowledge-for-reinforcement","title":"Transferring Knowledge for Reinforcement Learning in Contact-Rich Manipulation","date":"2022-09-19","arxiv_id":"2210.02891","repositories_listed":0,"syntology":null},{"url":null,"slug":"evolutionary-deep-reinforcement-learning","title":"Evolutionary Deep Reinforcement Learning Using Elite Buffer: A Novel Approach Towards DRL Combined with EA in Continuous Control Tasks","date":"2022-09-18","arxiv_id":"2209.08480","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-level-explanation-of-deep-reinforcement","title":"Multi-level Explanation of Deep Reinforcement Learning-based Scheduling","date":"2022-09-18","arxiv_id":"2209.09645","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-reinforcement-learning-with-4","title":"Offline Reinforcement Learning with Instrumental Variables in Confounded Markov Decision Processes","date":"2022-09-18","arxiv_id":"2209.08666","repositories_listed":0,"syntology":null},{"url":null,"slug":"intrinsically-motivated-reinforcement-2","title":"Intrinsically Motivated Reinforcement Learning based Recommendation with Counterfactual Data Augmentation","date":"2022-09-17","arxiv_id":"2209.08228","repositories_listed":0,"syntology":null},{"url":null,"slug":"ma2ql-a-minimalist-approach-to-fully","title":"MA2QL: A Minimalist Approach to Fully Decentralized Multi-Agent Reinforcement Learning","date":"2022-09-17","arxiv_id":"2209.08244","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-efficient-multi-agent-reinforcement","title":"Sample-Efficient Multi-Agent Reinforcement Learning with Demonstrations for Flocking Control","date":"2022-09-17","arxiv_id":"2209.08351","repositories_listed":0,"syntology":null},{"url":null,"slug":"sub-optimal-policy-aided-multi-agent","title":"Sub-optimal Policy Aided Multi-Agent Reinforcement Learning for Flocking Control","date":"2022-09-17","arxiv_id":"2209.08347","repositories_listed":0,"syntology":null},{"url":null,"slug":"conservative-dual-policy-optimization-for","title":"Conservative Dual Policy Optimization for Efficient Model-Based Reinforcement Learning","date":"2022-09-16","arxiv_id":"2209.07676","repositories_listed":0,"syntology":null},{"url":null,"slug":"neuromuscular-reinforcement-learning-to","title":"Neuromuscular Reinforcement Learning to Actuate Human Limbs through FES","date":"2022-09-16","arxiv_id":"2209.07849","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-industrial-hvac-systems-with","title":"Optimizing Industrial HVAC Systems with Hierarchical Reinforcement Learning","date":"2022-09-16","arxiv_id":"2209.08112","repositories_listed":0,"syntology":null},{"url":null,"slug":"trustworthy-reinforcement-learning-against","title":"Trustworthy Reinforcement Learning Against Intrinsic Vulnerabilities: Robustness, Safety, and Generalizability","date":"2022-09-16","arxiv_id":"2209.08025","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-offline-reinforcement-learning-help","title":"Can Offline Reinforcement Learning Help Natural Language Understanding?","date":"2022-09-15","arxiv_id":"2212.03864","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-task-1","title":"Deep Reinforcement Learning for Task Offloading in UAV-Aided Smart Farm Networks","date":"2022-09-15","arxiv_id":"2209.07367","repositories_listed":0,"syntology":null},{"url":null,"slug":"mean-field-approximation-of-cooperative","title":"Mean-Field Approximation of Cooperative Constrained Multi-Agent Reinforcement Learning (CMARL)","date":"2022-09-15","arxiv_id":"2209.07437","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixrts-toward-interpretable-multi-agent","title":"MIXRTs: Toward Interpretable Multi-Agent Reinforcement Learning via Mixing Recurrent Soft Decision Trees","date":"2022-09-15","arxiv_id":"2209.07225","repositories_listed":0,"syntology":null},{"url":null,"slug":"proapt-projection-of-apt-threats-with-deep","title":"ProAPT: Projection of APT Threats with Deep Reinforcement Learning","date":"2022-09-15","arxiv_id":"2209.07215","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-task-driven-robotic-swarm-control","title":"Scalable Task-Driven Robotic Swarm Control via Collision Avoidance and Learning Mean-Field Control","date":"2022-09-15","arxiv_id":"2209.07420","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-deep-neural-function","title":"Understanding Deep Neural Function Approximation in Reinforcement Learning via $ε$-Greedy Exploration","date":"2022-09-15","arxiv_id":"2209.07376","repositories_listed":0,"syntology":null},{"url":null,"slug":"analysis-of-reinforcement-learning-for","title":"Analysis of Reinforcement Learning for determining task replication in workflows","date":"2022-09-14","arxiv_id":"2209.13531","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributionally-robust-offline-reinforcement","title":"Distributionally Robust Offline Reinforcement Learning with Linear Function Approximation","date":"2022-09-14","arxiv_id":"2209.06620","repositories_listed":0,"syntology":null},{"url":null,"slug":"feature-rich-long-term-bitcoin-trading","title":"Feature-Rich Long-term Bitcoin Trading Assistant","date":"2022-09-14","arxiv_id":"2209.12664","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-constrained-reinforcement-learning","title":"Robust Constrained Reinforcement Learning","date":"2022-09-14","arxiv_id":"2209.06866","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-new-reinforcement-learning-framework-to","title":"A new Reinforcement Learning framework to discover natural flavor molecules","date":"2022-09-13","arxiv_id":"2209.05859","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-perception-applied-to-unmanned-aerial","title":"Active Perception Applied To Unmanned Aerial Vehicles Through Deep Reinforcement Learning","date":"2022-09-13","arxiv_id":"2209.06336","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-efficient-reinforcement-learning-and","title":"Data efficient reinforcement learning and adaptive optimal perimeter control of network traffic dynamics","date":"2022-09-13","arxiv_id":"2209.05726","repositories_listed":0,"syntology":null},{"url":null,"slug":"designing-biological-sequences-via-meta","title":"Designing Biological Sequences via Meta-Reinforcement Learning and Bayesian Optimization","date":"2022-09-13","arxiv_id":"2209.06259","repositories_listed":0,"syntology":null},{"url":null,"slug":"skip-training-for-multi-agent-reinforcement","title":"Skip Training for Multi-Agent Reinforcement Learning Controller for Industrial Wave Energy Converters","date":"2022-09-13","arxiv_id":"2209.05656","repositories_listed":0,"syntology":null},{"url":null,"slug":"unifying-causal-inference-and-reinforcement","title":"Unifying Causal Inference and Reinforcement Learning using Higher-Order Category Theory","date":"2022-09-13","arxiv_id":"2209.06262","repositories_listed":0,"syntology":null},{"url":null,"slug":"checklist-models-for-improved-output-fluency","title":"Checklist Models for Improved Output Fluency in Piano Fingering Prediction","date":"2022-09-12","arxiv_id":"2209.05622","repositories_listed":0,"syntology":null},{"url":null,"slug":"deterministic-sequencing-of-exploration-and","title":"Deterministic Sequencing of Exploration and Exploitation for Reinforcement Learning","date":"2022-09-12","arxiv_id":"2209.05408","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-sequential-information","title":"Self-supervised Sequential Information Bottleneck for Robust Exploration in Deep Reinforcement Learning","date":"2022-09-12","arxiv_id":"2209.05333","repositories_listed":0,"syntology":null},{"url":null,"slug":"pathfinding-in-random-partially-observable","title":"Pathfinding in Random Partially Observable Environments with Vision-Informed Deep Reinforcement Learning","date":"2022-09-11","arxiv_id":"2209.04801","repositories_listed":0,"syntology":null},{"url":null,"slug":"performance-driven-controller-tuning-via","title":"Performance-Driven Controller Tuning via Derivative-Free Reinforcement Learning","date":"2022-09-11","arxiv_id":"2209.04854","repositories_listed":0,"syntology":null},{"url":null,"slug":"cooperation-and-competition-flocking-with","title":"Cooperation and Competition: Flocking with Evolutionary Multi-Agent Reinforcement Learning","date":"2022-09-10","arxiv_id":"2209.04696","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-with-contrastive","title":"Safe Reinforcement Learning with Contrastive Risk Prediction","date":"2022-09-10","arxiv_id":"2209.09648","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-analysis-of-deep-reinforcement-learning","title":"An Analysis of Deep Reinforcement Learning Agents for Text-based Games","date":"2022-09-09","arxiv_id":"2209.04105","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixed-mathcal-h-2-mathcal-h-infty-lq-games","title":"Robust Policy Optimization in Continuous-time Mixed $\\mathcal{H}_2/\\mathcal{H}_\\infty$ Stochastic Control","date":"2022-09-09","arxiv_id":"2209.04477","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-large-population-systems-and","title":"A Survey on Large-Population Systems and Scalable Multi-Agent Reinforcement Learning","date":"2022-09-08","arxiv_id":"2209.03859","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybrid-supervised-and-reinforcement-learning","title":"Hybrid Supervised and Reinforcement Learning for the Design and Optimization of Nanophotonic Structures","date":"2022-09-08","arxiv_id":"2209.04447","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-mesh-generation-for-a-blade-passage","title":"Non-iterative generation of an optimal mesh for a blade passage using deep reinforcement learning","date":"2022-09-08","arxiv_id":"2209.05280","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-strategy-for","title":"A Deep Reinforcement Learning Strategy for UAV Autonomous Landing on a Platform","date":"2022-09-07","arxiv_id":"2209.02954","repositories_listed":0,"syntology":null},{"url":null,"slug":"concept-modulated-model-based-offline","title":"Concept-modulated model-based offline reinforcement learning for rapid generalization","date":"2022-09-07","arxiv_id":"2209.03207","repositories_listed":0,"syntology":null},{"url":null,"slug":"distilling-deep-rl-models-into-interpretable","title":"Distilling Deep RL Models Into Interpretable Neuro-Fuzzy Systems","date":"2022-09-07","arxiv_id":"2209.03357","repositories_listed":0,"syntology":null},{"url":null,"slug":"finite-time-error-bounds-for-greedy-gq","title":"Finite-Time Error Bounds for Greedy-GQ","date":"2022-09-06","arxiv_id":"2209.02555","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-assistive-robotics-with-deep","title":"Improving Assistive Robotics with Deep Reinforcement Learning","date":"2022-09-05","arxiv_id":"2209.02160","repositories_listed":0,"syntology":null},{"url":null,"slug":"natural-policy-gradients-in-reinforcement","title":"Natural Policy Gradients In Reinforcement Learning Explained","date":"2022-09-05","arxiv_id":"2209.01820","repositories_listed":0,"syntology":null},{"url":null,"slug":"prediction-based-decision-making-for","title":"Prediction Based Decision Making for Autonomous Highway Driving","date":"2022-09-05","arxiv_id":"2209.02106","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-optimised","title":"Reinforcement learning-based optimised control for tracking of nonlinear systems with adversarial attacks","date":"2022-09-05","arxiv_id":"2209.02165","repositories_listed":0,"syntology":null},{"url":null,"slug":"slatefree-a-model-free-decomposition-for","title":"SlateFree: a Model-Free Decomposition for Reinforcement Learning with Slate Actions","date":"2022-09-05","arxiv_id":"2209.01876","repositories_listed":0,"syntology":null},{"url":null,"slug":"variational-inference-for-model-free-and","title":"Variational Inference for Model-Free and Model-Based Reinforcement Learning","date":"2022-09-04","arxiv_id":"2209.01693","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-free-deep-reinforcement-learning-in","title":"Model-Free Deep Reinforcement Learning in Software-Defined Networks","date":"2022-09-03","arxiv_id":"2209.01490","repositories_listed":0,"syntology":null},{"url":null,"slug":"dialogue-evaluation-with-offline","title":"Dialogue Evaluation with Offline Reinforcement Learning","date":"2022-09-02","arxiv_id":"2209.00876","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-practical-communication-strategies","title":"Learning Practical Communication Strategies in Cooperative Multi-Agent Reinforcement Learning","date":"2022-09-02","arxiv_id":"2209.01288","repositories_listed":0,"syntology":null},{"url":null,"slug":"semi-centralised-multi-agent-reinforcement","title":"Taming Multi-Agent Reinforcement Learning with Estimator Variance Reduction","date":"2022-09-02","arxiv_id":"2209.01054","repositories_listed":0,"syntology":null},{"url":null,"slug":"targf-learning-target-gradient-field-for","title":"TarGF: Learning Target Gradient Field to Rearrange Objects without Explicit Goal Specification","date":"2022-09-02","arxiv_id":"2209.00853","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-technique-to-create-weaker-abstract-board","title":"A Technique to Create Weaker Abstract Board Game Agents via Reinforcement Learning","date":"2022-09-01","arxiv_id":"2209.00711","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-quantum-1","title":"Deep reinforcement learning for quantum multiparameter estimation","date":"2022-09-01","arxiv_id":"2209.00671","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamics-adaptive-continual-reinforcement","title":"Dynamics-Adaptive Continual Reinforcement Learning via Progressive Contextualization","date":"2022-09-01","arxiv_id":"2209.00347","repositories_listed":0,"syntology":null},{"url":null,"slug":"metatrader-an-reinforcement-learning-approach","title":"MetaTrader: An Reinforcement Learning Approach Integrating Diverse Policies for Portfolio Optimization","date":"2022-09-01","arxiv_id":"2210.01774","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-stabilizing-reinforcement-learning-approach","title":"A stabilizing reinforcement learning approach for sampled systems with partially unknown models","date":"2022-08-31","arxiv_id":"2208.14714","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-anomaly-detection-and-search-via","title":"Deep Anomaly Detection and Search via Reinforcement Learning","date":"2022-08-31","arxiv_id":"2208.14834","repositories_listed":0,"syntology":null},{"url":null,"slug":"transmit-power-control-for-indoor-small-cells","title":"Transmit Power Control for Indoor Small Cells: A Method Based on Federated Reinforcement Learning","date":"2022-08-31","arxiv_id":"2209.13536","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-further-exploration-of-deep-multi-agent","title":"A further exploration of deep Multi-Agent Reinforcement Learning with Hybrid Action Space","date":"2022-08-30","arxiv_id":"2208.14447","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-analysis-of-abstracted-model-based-1","title":"An Analysis of Model-Based Reinforcement Learning From Abstracted Observations","date":"2022-08-30","arxiv_id":"2208.14407","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-ensembles-of-reinforcement","title":"Distributed Ensembles of Reinforcement Learning Agents for Electricity Control","date":"2022-08-30","arxiv_id":"2208.14338","repositories_listed":0,"syntology":null},{"url":null,"slug":"evolutionary-deep-reinforcement-learning-for","title":"Evolutionary Deep Reinforcement Learning for Dynamic Slice Management in O-RAN","date":"2022-08-30","arxiv_id":"2208.14394","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-reinforcement-learning-with-sindy","title":"Model-Based Reinforcement Learning with SINDy","date":"2022-08-30","arxiv_id":"2208.14501","repositories_listed":0,"syntology":null}],"record_sha256":"3ce8f893551917ba313325166b2205f2b0d4ce6576682d389f8c62eff2a6b078","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}