{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/110","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":110,"pages_in_order":152,"rows_per_page":100,"rows":[10901,11000],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/109","next":"/task/reinforcement-learning-1/papers/111","papers":[{"url":null,"slug":"stable-deep-reinforcement-learning-method-by","title":"Stable deep reinforcement learning method by predicting uncertainty in rewards as a subtask","date":"2021-01-18","arxiv_id":"2101.06906","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-safe-hierarchical-planning-framework-for","title":"A Safe Hierarchical Planning Framework for Complex Driving Scenarios based on Reinforcement Learning","date":"2021-01-17","arxiv_id":"2101.06778","repositories_listed":0,"syntology":null},{"url":null,"slug":"affordance-based-reinforcement-learning-for","title":"Affordance-based Reinforcement Learning for Urban Driving","date":"2021-01-15","arxiv_id":"2101.05970","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-haptic-shared","title":"Deep Reinforcement Learning for Haptic Shared Control in Unknown Tasks","date":"2021-01-15","arxiv_id":"2101.06227","repositories_listed":0,"syntology":null},{"url":null,"slug":"empirical-evaluation-of-supervision-signals","title":"Empirical Evaluation of Supervision Signals for Style Transfer Models","date":"2021-01-15","arxiv_id":"2101.06172","repositories_listed":0,"syntology":null},{"url":null,"slug":"local-navigation-and-docking-of-an-autonomous","title":"Local Navigation and Docking of an Autonomous Robot Mower using Reinforcement Learning and Computer Vision","date":"2021-01-15","arxiv_id":"2101.06248","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-recommender-1","title":"Reinforcement learning based recommender systems: A survey","date":"2021-01-15","arxiv_id":"2101.06286","repositories_listed":0,"syntology":null},{"url":null,"slug":"robusta-robust-automl-for-feature-selection","title":"Robusta: Robust AutoML for Feature Selection via Reinforcement Learning","date":"2021-01-15","arxiv_id":"2101.05950","repositories_listed":0,"syntology":null},{"url":null,"slug":"stochastic-learning-approach-to-binary","title":"Stochastic Learning Approach to Binary Optimization for Optimal Design of Experiments","date":"2021-01-15","arxiv_id":"2101.05958","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-and-fast-adaptation-for-grid","title":"Learning and Fast Adaptation for Grid Emergency Control via Deep Meta Reinforcement Learning","date":"2021-01-13","arxiv_id":"2101.05317","repositories_listed":0,"syntology":null},{"url":null,"slug":"automated-synthesis-of-steady-state","title":"Automated Synthesis of Steady-State Continuous Processes using Reinforcement Learning","date":"2021-01-12","arxiv_id":"2101.04422","repositories_listed":0,"syntology":null},{"url":null,"slug":"linear-representation-meta-reinforcement-1","title":"Linear Representation Meta-Reinforcement Learning for Instant Adaptation","date":"2021-01-12","arxiv_id":"2101.04750","repositories_listed":0,"syntology":null},{"url":null,"slug":"queue-learning-a-reinforcement-learning","title":"Queue-Learning: A Reinforcement Learning Approach for Providing Quality of Service","date":"2021-01-12","arxiv_id":"2101.04627","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-interactive-bayesian-reinforcement","title":"Deep Interactive Bayesian Reinforcement Learning via Meta-Learning","date":"2021-01-11","arxiv_id":"2101.03864","repositories_listed":0,"syntology":null},{"url":null,"slug":"first-order-problem-solving-through-neural","title":"First-Order Problem Solving through Neural MCTS based Reinforcement Learning","date":"2021-01-11","arxiv_id":"2101.04167","repositories_listed":0,"syntology":null},{"url":null,"slug":"independent-policy-gradient-methods-for-1","title":"Independent Policy Gradient Methods for Competitive Reinforcement Learning","date":"2021-01-11","arxiv_id":"2101.04233","repositories_listed":0,"syntology":null},{"url":null,"slug":"identifying-decision-points-for-safe-and","title":"Identifying Decision Points for Safe and Interpretable Reinforcement Learning in Hypotension Treatment","date":"2021-01-09","arxiv_id":"2101.03309","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-and-scalable-routing-with-multi-agent","title":"Robust and Scalable Routing with Multi-Agent Deep Reinforcement Learning for MANETs","date":"2021-01-09","arxiv_id":"2101.03273","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-coupled-deep-q-learning-for","title":"Safe Coupled Deep Q-Learning for Recommendation Systems","date":"2021-01-08","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"active-screening-for-recurrent-diseases-a","title":"Active Screening for Recurrent Diseases: A Reinforcement Learning Approach","date":"2021-01-07","arxiv_id":"2101.02766","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-adaptive-multi-agent-physical-layer","title":"An Adaptive Multi-Agent Physical Layer Security Framework for Cognitive Cyber-Physical Systems","date":"2021-01-07","arxiv_id":"2101.02446","repositories_listed":0,"syntology":null},{"url":null,"slug":"coachnet-an-adversarial-sampling-approach-for","title":"CoachNet: An Adversarial Sampling Approach for Reinforcement Learning","date":"2021-01-07","arxiv_id":"2101.02649","repositories_listed":0,"syntology":null},{"url":null,"slug":"coding-for-distributed-multi-agent","title":"Coding for Distributed Multi-Agent Reinforcement Learning","date":"2021-01-07","arxiv_id":"2101.02308","repositories_listed":0,"syntology":null},{"url":null,"slug":"qrrt-quality-biased-incremental-rrt-for","title":"qRRT: Quality-Biased Incremental RRT for Optimal Motion Planning in Non-Holonomic Systems","date":"2021-01-07","arxiv_id":"2101.02635","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-quantum","title":"Deep Reinforcement Learning with Quantum-inspired Experience Replay","date":"2021-01-06","arxiv_id":"2101.02034","repositories_listed":0,"syntology":null},{"url":null,"slug":"geometric-entropic-exploration","title":"Geometric Entropic Exploration","date":"2021-01-06","arxiv_id":"2101.02055","repositories_listed":0,"syntology":null},{"url":null,"slug":"off-policy-meta-reinforcement-learning-based","title":"Off-Policy Meta-Reinforcement Learning Based on Feature Embedding Spaces","date":"2021-01-06","arxiv_id":"2101.01883","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-efficient-reinforcement-learning-4","title":"Provably Efficient Reinforcement Learning with Linear Function Approximation Under Adaptivity Constraints","date":"2021-01-06","arxiv_id":"2101.02195","repositories_listed":0,"syntology":null},{"url":null,"slug":"smoothed-functional-based-gradient-algorithms","title":"Smoothed functional-based gradient algorithms for off-policy reinforcement learning: A non-asymptotic viewpoint","date":"2021-01-06","arxiv_id":"2101.02137","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-a-curriculum-approach-to-reinforcement","title":"An A* Curriculum Approach to Reinforcement Learning for RGBD Indoor Robot Navigation","date":"2021-01-05","arxiv_id":"2101.01774","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhanced-audit-techniques-empowered-by-the","title":"Enhanced Audit Techniques Empowered by the Reinforcement Learning Pertaining to IFRS 16 Lease","date":"2021-01-05","arxiv_id":"2101.05633","repositories_listed":0,"syntology":null},{"url":null,"slug":"derivative-free-policy-optimization-for-risk","title":"Derivative-Free Policy Optimization for Linear Risk-Sensitive and Robust Control Design: Implicit Regularization and Sample Complexity","date":"2021-01-04","arxiv_id":"2101.01041","repositories_listed":0,"syntology":null},{"url":null,"slug":"markov-chain-monte-carlo-policy-optimization","title":"Markov Chain Monte Carlo Policy Optimization","date":"2021-01-04","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"enhanced-pub-sub-communications-for-massive","title":"Enhanced Pub/Sub Communications for Massive IoT Traffic with SARSA Reinforcement Learning","date":"2021-01-03","arxiv_id":"2101.00687","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-joint-learning-and-communication-framework","title":"Effective Communications: A Joint Learning and Communication Framework for Multi-Agent Reinforcement Learning over Noisy Channels","date":"2021-01-02","arxiv_id":"2101.10369","repositories_listed":0,"syntology":null},{"url":null,"slug":"context-aware-safe-reinforcement-learning-for","title":"Context-Aware Safe Reinforcement Learning for Non-Stationary Environments","date":"2021-01-02","arxiv_id":"2101.00531","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-flexibility-design-1","title":"Reinforcement Learning for Flexibility Design Problems","date":"2021-01-02","arxiv_id":"2101.00355","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reduction-approach-to-constrained","title":"A Reduction Approach to Constrained Reinforcement Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-framework-for-time-1","title":"A REINFORCEMENT LEARNING FRAMEWORK FOR TIME DEPENDENT CAUSAL EFFECTS EVALUATION IN A/B TESTING","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-robust-fuel-optimization-strategy-for","title":"A Robust Fuel Optimization Strategy For Hybrid Electric Vehicles: A Deep Reinforcement Learning Based Continuous Time Design Approach","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-simple-sparse-denoising-layer-for-robust","title":"A Simple Sparse Denoising Layer for Robust Deep Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-deep-reinforcement-learning-for","title":"A Survey on Deep Reinforcement Learning for Audio-Based Applications","date":"2021-01-01","arxiv_id":"2101.00240","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-learning-rates-for-multi-agent","title":"Adaptive Learning Rates for Multi-Agent Reinforcement Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-multi-model-fusion-learning-for","title":"Adaptive Multi-model Fusion Learning for Sparse-Reward Reinforcement Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"addressing-distribution-shift-in-online","title":"Addressing Distribution Shift in Online Reinforcement Learning with Offline Datasets","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"addressing-extrapolation-error-in-deep","title":"Addressing Extrapolation Error in Deep Offline Reinforcement Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"alpha-dag-a-reinforcement-learning-based","title":"Alpha-DAG: a reinforcement learning based algorithm to learn Directed Acyclic Graphs","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"an-examination-of-preference-based","title":"An Examination of Preference-based Reinforcement Learning for Treatment Recommendation","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"approximating-pareto-frontier-through","title":"Approximating Pareto Frontier through Bayesian-optimization-directed Robust Multi-objective Reinforcement Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"aspect-based-sentiment-classification-via","title":"Aspect-based Sentiment Classification via Reinforcement Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-driven-robotic-manipulation","title":"Attention-driven Robotic Manipulation","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"average-reward-reinforcement-learning-with-1","title":"Average Reward Reinforcement Learning with Monotonic Policy Improvement","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"batch-reinforcement-learning-through","title":"Batch Reinforcement Learning Through Continuation Method","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-multi-agent-deep-reinforcement","title":"Benchmarking Multi-Agent Deep Reinforcement Learning Algorithms","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"bounded-myopic-adversaries-for-deep","title":"Bounded Myopic Adversaries for Deep Reinforcement Learning Agents","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"brac-going-deeper-with-behavior-regularized","title":"BRAC+: Going Deeper with Behavior Regularized Offline Reinforcement Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"cat-sac-soft-actor-critic-with-curiosity","title":"CAT-SAC: Soft Actor-Critic with Curiosity-Aware Entropy Temperature","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"combining-imitation-and-reinforcement","title":"Combining Imitation and Reinforcement Learning with Free Energy Principle","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"communication-in-multi-agent-reinforcement","title":"Communication in Multi-Agent Reinforcement Learning: Intention Sharing","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"compute-and-memory-efficient-reinforcement","title":"Compute- and Memory-Efficient Reinforcement Learning with Latent Experience Replay","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"constrained-reinforcement-learning-with","title":"Constrained Reinforcement Learning With Learned Constraints","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"coordinated-multi-agent-exploration-using","title":"Coordinated Multi-Agent Exploration Using Shared Goals","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-state-self-constraint-for-feature","title":"Cross-State Self-Constraint for Feature Generalization in Deep Reinforcement Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"daylight-assessing-generalization-skills-of","title":"Daylight: Assessing Generalization Skills of Deep Reinforcement Learning Agents","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-coherent-exploration-for-continuous","title":"Deep Coherent Exploration For Continuous Control","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-anti","title":"Deep Reinforcement Learning-based Anti-jamming Power Allocation in a Two-cell NOMA Network","date":"2021-01-01","arxiv_id":"2101.00270","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-adaptive","title":"Deep Reinforcement Learning With Adaptive Combined Critics","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"discrete-predictive-representation-for-long","title":"Discrete Predictive Representation for Long-horizon Planning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"distributional-reinforcement-learning-for-2","title":"Distributional Reinforcement Learning for Risk-Sensitive Policies","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"divide-and-conquer-monte-carlo-tree-search-1","title":"Divide-and-Conquer Monte Carlo Tree Search","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"entropic-risk-sensitive-reinforcement","title":"Entropic Risk-Sensitive Reinforcement Learning: A Meta Regret Framework with Function Approximation","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"error-controlled-actor-critic-method-to","title":"Error Controlled Actor-Critic Method to Reinforcement Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"explainable-reinforcement-learning-through","title":"Explainable Reinforcement Learning Through Goal-Based Explanations","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"explicit-pareto-front-optimization-for","title":"Explicit Pareto Front Optimization for Constrained Reinforcement Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"explore-with-dynamic-map-graph-structured","title":"Explore with Dynamic Map: Graph Structured Reinforcement Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-transferability-of-perturbations-in","title":"Exploring Transferability of Perturbations in Deep Reinforcement Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"factored-action-spaces-in-deep-reinforcement","title":"Factored Action Spaces in Deep Reinforcement Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"factoredrl-leveraging-factored-graphs-for","title":"FactoredRL: Leveraging Factored Graphs for Deep Reinforcement Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"fine-tuning-offline-reinforcement-learning","title":"Fine-Tuning Offline Reinforcement Learning with Model-Based Policy Optimization","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"fsv-learning-to-factorize-soft-value-function","title":"FSV: Learning to Factorize Soft Value Function for Cooperative Multi-Agent Reinforcement Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"genetic-soft-updates-for-policy-evolution-in","title":"Genetic Soft Updates for Policy Evolution in Deep Reinforcement Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"grounding-language-to-entities-for","title":"Grounding Language to Entities for Generalization in Reinforcement Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"guiding-representation-learning-in-deep","title":"Guiding Representation Learning in Deep Generative Models with Policy Gradients","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"hellinger-distance-constrained-regression","title":"Hellinger Distance Constrained Regression","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"hindsight-curriculum-generation-based-multi","title":"Hindsight Curriculum Generation Based Multi-Goal Experience Replay","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-learning-to-branch-via","title":"Improving Learning to Branch via Reinforcement Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"incremental-policy-gradients-for-online","title":"Incremental Policy Gradients for Online Reinforcement Learning Control","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-meta-reinforcement-learning","title":"Interpretable Meta-Reinforcement Learning with Actor-Critic Method","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-reinforcement-learning-with-1","title":"Interpretable Reinforcement Learning With Neural Symbolic Logic","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"intrinsically-guided-exploration-in-meta","title":"Intrinsically Guided Exploration in Meta Reinforcement Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"invariant-representations-for-reinforcement","title":"Invariant Representations for Reinforcement Learning without Reconstruction","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"inverse-reinforcement-learning-for-autonomous","title":"Inverse reinforcement learning for autonomous navigation via differentiable semantic mapping and planning","date":"2021-01-01","arxiv_id":"2101.00186","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-a-transferable-scheduling-policy-for","title":"Learning a Transferable Scheduling Policy for Various Vehicle Routing Problems based on Graph-centric Representation Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-active-learning-in-the-batch-mode","title":"Learning Active Learning in the Batch-Mode Setup with Ensembles of Active Learning Agents","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-efficient-planning-based-rewards-for","title":"Learning Efficient Planning-based Rewards for Imitation Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-from-demonstrations-with-energy","title":"Learning from Demonstrations with Energy based Generative Adversarial Imitation Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-latent-landmarks-for-generalizable","title":"Learning Latent Landmarks for Generalizable Planning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-predictive-communication-by","title":"Learning Predictive Communication by Imagination in Networked System Control","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-safe-policies-with-cost-sensitive","title":"Learning Safe Policies with Cost-sensitive Advantage Estimation","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-communicate-through-imagination","title":"Learning to communicate through imagination with model-based deep multi-agent reinforcement learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null}],"record_sha256":"e88d97b74de5f5cdfa49fac1586dab1dffb1a5af8e3c44593d685f2f72583422","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}