{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/122","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":122,"pages_in_order":152,"rows_per_page":100,"rows":[12101,12200],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/121","next":"/task/reinforcement-learning-1/papers/123","papers":[{"url":null,"slug":"toma-topological-map-abstraction-for","title":"TOMA: Topological Map Abstraction for Reinforcement Learning","date":"2020-05-11","arxiv_id":"2005.06061","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-deep-neuroevolution-on","title":"Accelerating Deep Neuroevolution on Distributed FPGAs for Reinforcement Learning Problems","date":"2020-05-10","arxiv_id":"2005.04536","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-fpga-based-on-device-reinforcement","title":"An FPGA-Based On-Device Reinforcement Learning Approach using Online Sequential Learning","date":"2020-05-10","arxiv_id":"2005.04646","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-pid-and-antiwindup-control-design-as","title":"Optimal PID and Antiwindup Control Design as a Reinforcement Learning Problem","date":"2020-05-10","arxiv_id":"2005.04539","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-beamforming-for","title":"A Reinforcement Learning based approach for Multi-target Detection in Massive MIMO radar","date":"2020-05-10","arxiv_id":"2005.04708","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-design-of-linear","title":"Reinforcement Learning based Design of Linear Fixed Structure Controllers","date":"2020-05-10","arxiv_id":"2005.04537","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-thermostatically","title":"Reinforcement Learning for Thermostatically Controlled Loads Control using Modelica and Python","date":"2020-05-09","arxiv_id":"2005.04444","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-deep-reinforcement-learning-ready-for","title":"Is Deep Reinforcement Learning Ready for Practical Applications in Healthcare? A Sensitivity Analysis of Duel-DDQN for Hemodynamic Management in Sepsis Patients","date":"2020-05-08","arxiv_id":"2005.04301","repositories_listed":0,"syntology":null},{"url":null,"slug":"synthesizing-safe-policies-under","title":"Synthesizing Safe Policies under Probabilistic Constraints with Reinforcement Learning and Bayesian Model Checking","date":"2020-05-08","arxiv_id":"2005.03898","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-dialog-policy-learning-with","title":"Adaptive Dialog Policy Learning with Hindsight and User Modeling","date":"2020-05-07","arxiv_id":"2005.03299","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-feedback-graphs","title":"Reinforcement Learning with Feedback Graphs","date":"2020-05-07","arxiv_id":"2005.03789","repositories_listed":0,"syntology":null},{"url":null,"slug":"robotic-arm-control-and-task-training-through","title":"Robotic Arm Control and Task Training through Deep Reinforcement Learning","date":"2020-05-06","arxiv_id":"2005.02632","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-through-meta","title":"Safe Reinforcement Learning through Meta-learned Instincts","date":"2020-05-06","arxiv_id":"2005.03233","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-dialog-management-recent-advances","title":"A Survey on Dialog Management: Recent Advances and Challenges","date":"2020-05-05","arxiv_id":"2005.02233","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalized-planning-with-deep-reinforcement","title":"Generalized Planning With Deep Reinforcement Learning","date":"2020-05-05","arxiv_id":"2005.02305","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-uav-autonomous","title":"Reinforcement Learning for UAV Autonomous Navigation, Mapping and Target Detection","date":"2020-05-05","arxiv_id":"2005.05057","repositories_listed":0,"syntology":null},{"url":null,"slug":"formal-policy-synthesis-for-continuous-space","title":"Formal Policy Synthesis for Continuous-Space Systems via Reinforcement Learning","date":"2020-05-04","arxiv_id":"2005.01319","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalized-reinforcement-meta-learning-for","title":"Generalized Reinforcement Meta Learning for Few-Shot Optimization","date":"2020-05-04","arxiv_id":"2005.01246","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-decomposition-of-nonlinear","title":"Hierarchical Decomposition of Nonlinear Dynamics and Control for System Identification and Policy Distillation","date":"2020-05-04","arxiv_id":"2005.01432","repositories_listed":0,"syntology":null},{"url":null,"slug":"multiagent-value-iteration-algorithms-in","title":"Multiagent Value Iteration Algorithms in Dynamic Programming and Reinforcement Learning","date":"2020-05-04","arxiv_id":"2005.01627","repositories_listed":0,"syntology":null},{"url":null,"slug":"noise-pollution-in-hospital-readmission","title":"Noise Pollution in Hospital Readmission Prediction: Long Document Classification with Reinforcement Learning","date":"2020-05-04","arxiv_id":"2005.01259","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-constrained-interactive-recommendation","title":"Reward Constrained Interactive Recommendation with Natural Language Feedback","date":"2020-05-04","arxiv_id":"2005.01618","repositories_listed":0,"syntology":null},{"url":null,"slug":"setting-up-experimental-bell-test-with","title":"Setting up experimental Bell test with reinforcement learning","date":"2020-05-04","arxiv_id":"2005.01697","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-decentralized","title":"Multi-agent Reinforcement Learning for Decentralized Stable Matching","date":"2020-05-03","arxiv_id":"2005.01117","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-intelligent-1","title":"Deep Reinforcement Learning for Intelligent Transportation Systems: A Survey","date":"2020-05-02","arxiv_id":"2005.00935","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-text-based-reinforcement-learning","title":"Enhancing Text-based Reinforcement Learning Agents with Commonsense Knowledge","date":"2020-05-02","arxiv_id":"2005.00811","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-beam-association-for-high-mobility","title":"Optimal Beam Association for High Mobility mmWave Vehicular Networks: Lightweight Parallel Reinforcement Learning Approach","date":"2020-05-02","arxiv_id":"2005.00694","repositories_listed":0,"syntology":null},{"url":null,"slug":"amrl-aggregated-memory-for-reinforcement","title":"AMRL: Aggregated Memory For Reinforcement Learning","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"episodic-reinforcement-learning-with","title":"Episodic Reinforcement Learning with Associative Memory","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"exploration-in-reinforcement-learning-with","title":"Exploration in Reinforcement Learning with Deep Covering Options","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-robustness-via-risk-averse","title":"Improving Robustness via Risk Averse Distributional Reinforcement Learning","date":"2020-05-01","arxiv_id":"2005.00585","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-long-horizon-reinforcement-learning-more","title":"Is Long Horizon Reinforcement Learning More Difficult Than Short Horizon Reinforcement Learning?","date":"2020-05-01","arxiv_id":"2005.00527","repositories_listed":0,"syntology":null},{"url":null,"slug":"keep-doing-what-worked-behavior-modelling","title":"Keep Doing What Worked: Behavior Modelling Priors for Offline Reinforcement Learning","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-efficient-parameter-server","title":"Learning Efficient Parameter Server Synchronization Policies for Distributed SGD","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-heuristics-for-quantified-boolean","title":"Learning Heuristics for Quantified Boolean Formulas through Reinforcement Learning","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-the-arrow-of-time-for-problems-in","title":"Learning the Arrow of Time for Problems in Reinforcement Learning","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-reinforcement-learning-for","title":"Model-based reinforcement learning for biological sequence design","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-reinforcement-learning-for-atari-1","title":"Model Based Reinforcement Learning for Atari","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"posterior-sampling-for-multi-agent","title":"Posterior sampling for multi-agent reinforcement learning: solving extensive games with imperfect information","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"synthesizing-programmatic-policies-that","title":"Synthesizing Programmatic Policies that Inductively Generalize","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-ingredients-of-real-world-robotic","title":"The Ingredients of Real World Robotic Reinforcement Learning","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-evaluating-robustness-of-deep","title":"Toward Evaluating Robustness of Deep Reinforcement Learning with Continuous Control","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"bootstrap-latent-predictive-representations","title":"Bootstrap Latent-Predictive Representations for Multitask Reinforcement Learning","date":"2020-04-30","arxiv_id":"2004.14646","repositories_listed":0,"syntology":null},{"url":null,"slug":"breaking-global-barriers-in-parallel","title":"Breaking (Global) Barriers in Parallel Stochastic Optimization with Wait-Avoiding Group Averaging","date":"2020-04-30","arxiv_id":"2005.00124","repositories_listed":0,"syntology":null},{"url":null,"slug":"delay-aware-resource-allocation-in-fog","title":"Delay-aware Resource Allocation in Fog-assisted IoT Networks Through Reinforcement Learning","date":"2020-04-30","arxiv_id":"2005.04097","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributional-soft-actor-critic-for-risk","title":"DSAC: Distributional Soft Actor Critic for Risk-Sensitive Reinforcement Learning","date":"2020-04-30","arxiv_id":"2004.14547","repositories_listed":0,"syntology":null},{"url":null,"slug":"gcn-rl-circuit-designer-transferable","title":"GCN-RL Circuit Designer: Transferable Transistor Sizing with Graph Neural Networks and Reinforcement Learning","date":"2020-04-30","arxiv_id":"2005.00406","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-persona-consistent-dialogue","title":"Improving Factual Consistency Between a Response and Persona Facts","date":"2020-04-30","arxiv_id":"2005.00036","repositories_listed":0,"syntology":null},{"url":null,"slug":"out-of-the-box-channel-pruned-networks","title":"Out-of-the-box channel pruned networks","date":"2020-04-30","arxiv_id":"2004.14584","repositories_listed":0,"syntology":null},{"url":null,"slug":"plan-space-state-embeddings-for-improved","title":"Plan-Space State Embeddings for Improved Reinforcement Learning","date":"2020-04-30","arxiv_id":"2004.14567","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-of-minimalist-grammars","title":"Reinforcement learning of minimalist grammars","date":"2020-04-30","arxiv_id":"2005.00359","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-embodied-scene-description","title":"Towards Embodied Scene Description","date":"2020-04-30","arxiv_id":"2004.14638","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-learning-of-kb-queries-in-task","title":"Unsupervised Learning of KB Queries in Task-Oriented Dialogs","date":"2020-04-30","arxiv_id":"2005.00123","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-for-robotic","title":"Meta-Reinforcement Learning for Robotic Industrial Insertion Tasks","date":"2020-04-29","arxiv_id":"2004.14404","repositories_listed":0,"syntology":null},{"url":null,"slug":"molecular-design-in-synthetically-accessible","title":"Molecular Design in Synthetically Accessible Chemical Space via Deep Reinforcement Learning","date":"2020-04-29","arxiv_id":"2004.14308","repositories_listed":0,"syntology":null},{"url":null,"slug":"reduced-dimensional-reinforcement-learning","title":"Reduced-Dimensional Reinforcement Learning Control using Singular Perturbation Approximations","date":"2020-04-29","arxiv_id":"2004.14501","repositories_listed":0,"syntology":null},{"url":null,"slug":"whittle-index-based-q-learning-for-restless","title":"Whittle index based Q-learning for restless bandits with average reward","date":"2020-04-29","arxiv_id":"2004.14427","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-sample-efficiency-and-multi-agent","title":"Improving Sample Efficiency and Multi-Agent Communication in RL-based Train Rescheduling","date":"2020-04-28","arxiv_id":"2004.13439","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-immersion-of-directed-multi-graphs-in","title":"The Immersion of Directed Multi-graphs in Embedding Fields. Generalisations","date":"2020-04-28","arxiv_id":"2004.13384","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-model-selection-in-photonic","title":"Adaptive model selection in photonic reservoir computing by reinforcement learning","date":"2020-04-27","arxiv_id":"2004.12575","repositories_listed":0,"syntology":null},{"url":null,"slug":"age-aware-status-update-control-for-energy","title":"Age-Aware Status Update Control for Energy Harvesting IoT Sensors via Reinforcement Learning","date":"2020-04-27","arxiv_id":"2004.12684","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-we-learn-heuristics-for-graphical-model","title":"Can We Learn Heuristics For Graphical Model Inference Using Reinforcement Learning?","date":"2020-04-27","arxiv_id":"2005.01508","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-ingredients-of-real-world-robotic-1","title":"The Ingredients of Real-World Robotic Reinforcement Learning","date":"2020-04-27","arxiv_id":"2004.12570","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-state-aggregation-approach-for-solving","title":"A State Aggregation Approach for Solving Knapsack Problem with Deep Reinforcement Learning","date":"2020-04-25","arxiv_id":"2004.12117","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-low-bit-hybrid-quantization-of","title":"Automatic low-bit hybrid quantization of neural networks through meta learning","date":"2020-04-24","arxiv_id":"2004.11506","repositories_listed":0,"syntology":null},{"url":null,"slug":"pbcs-efficient-exploration-and-exploitation","title":"PBCS : Efficient Exploration and Exploitation Using a Synergy between Reinforcement Learning and Motion Planning","date":"2020-04-24","arxiv_id":"2004.11667","repositories_listed":0,"syntology":null},{"url":null,"slug":"cooperative-perception-with-deep","title":"Cooperative Perception with Deep Reinforcement Learning for Connected Vehicles","date":"2020-04-23","arxiv_id":"2004.10927","repositories_listed":0,"syntology":null},{"url":null,"slug":"divide-and-conquer-monte-carlo-tree-search","title":"Divide-and-Conquer Monte Carlo Tree Search For Goal-Directed Planning","date":"2020-04-23","arxiv_id":"2004.11410","repositories_listed":0,"syntology":null},{"url":null,"slug":"guided-dyna-q-for-mobile-robot-exploration","title":"Guiding Robot Exploration in Reinforcement Learning via Automated Planning","date":"2020-04-23","arxiv_id":"2004.11456","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-dialog-policies-from-weak","title":"Learning Dialog Policies from Weak Demonstrations","date":"2020-04-23","arxiv_id":"2004.11054","repositories_listed":0,"syntology":null},{"url":null,"slug":"autoeg-automated-experience-grafting-for-off","title":"AutoEG: Automated Experience Grafting for Off-Policy Deep Reinforcement Learning","date":"2020-04-22","arxiv_id":"2004.10698","repositories_listed":0,"syntology":null},{"url":null,"slug":"flexible-and-efficient-long-range-planning-1","title":"Flexible and Efficient Long-Range Planning Through Curious Exploration","date":"2020-04-22","arxiv_id":"2004.10876","repositories_listed":0,"syntology":null},{"url":null,"slug":"sequential-anomaly-detection-using-inverse","title":"Sequential Anomaly Detection using Inverse Reinforcement Learning","date":"2020-04-22","arxiv_id":"2004.10398","repositories_listed":0,"syntology":null},{"url":null,"slug":"almost-optimal-model-free-reinforcement","title":"Almost Optimal Model-Free Reinforcement Learning via Reference-Advantage Decomposition","date":"2020-04-21","arxiv_id":"2004.10019","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-adaptation-for-end-to-end-vision","title":"Never Stop Learning: The Effectiveness of Fine-Tuning in Robotic Reinforcement Learning","date":"2020-04-21","arxiv_id":"2004.10190","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-to-optimize-the","title":"Reinforcement Learning to Optimize the Logistics Distribution Routes of Unmanned Aerial Vehicle","date":"2020-04-21","arxiv_id":"2004.09864","repositories_listed":0,"syntology":null},{"url":null,"slug":"sibre-self-improvement-based-rewards-for","title":"SIBRE: Self Improvement Based REwards for Adaptive Feedback in Reinforcement Learning","date":"2020-04-21","arxiv_id":"2004.09846","repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-routing-track-assignment-detailed","title":"Attention Routing: track-assignment detailed routing using attention-based reinforcement learning","date":"2020-04-20","arxiv_id":"2004.09473","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-driven-learning-and-load-ensemble","title":"Data-Driven Learning and Load Ensemble Control","date":"2020-04-20","arxiv_id":"2004.09675","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-as-reinforcement-applying-principles","title":"Learning as Reinforcement: Applying Principles of Neuroscience for More General Reinforcement Learning Agents","date":"2020-04-20","arxiv_id":"2004.09043","repositories_listed":0,"syntology":null},{"url":null,"slug":"tightening-exploration-in-upper-confidence","title":"Tightening Exploration in Upper Confidence Reinforcement Learning","date":"2020-04-20","arxiv_id":"2004.09656","repositories_listed":0,"syntology":null},{"url":null,"slug":"intention-propagation-for-multi-agent","title":"Variational Policy Propagation for Multi-agent Reinforcement Learning","date":"2020-04-19","arxiv_id":"2004.08883","repositories_listed":0,"syntology":null},{"url":null,"slug":"superkernel-neural-architecture-search-for","title":"Superkernel Neural Architecture Search for Image Denoising","date":"2020-04-19","arxiv_id":"2004.08870","repositories_listed":0,"syntology":null},{"url":null,"slug":"macro-action-based-deep-multi-agent","title":"Macro-Action-Based Deep Multi-Agent Reinforcement Learning","date":"2020-04-18","arxiv_id":"2004.08646","repositories_listed":0,"syntology":null},{"url":null,"slug":"modeling-survival-in-model-based","title":"Modeling Survival in model-based Reinforcement Learning","date":"2020-04-18","arxiv_id":"2004.08648","repositories_listed":0,"syntology":null},{"url":null,"slug":"time-adaptive-reinforcement-learning","title":"Time Adaptive Reinforcement Learning","date":"2020-04-18","arxiv_id":"2004.08600","repositories_listed":0,"syntology":null},{"url":null,"slug":"approximate-inverse-reinforcement-learning","title":"Approximate Inverse Reinforcement Learning from Vision-based Imitation Learning","date":"2020-04-17","arxiv_id":"2004.08051","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-adaptive-1","title":"Deep Reinforcement Learning for Adaptive Learning Systems","date":"2020-04-17","arxiv_id":"2004.08410","repositories_listed":0,"syntology":null},{"url":null,"slug":"f2a2-flexible-fully-decentralized-approximate","title":"F2A2: Flexible Fully-decentralized Approximate Actor-critic for Cooperative Multi-agent Reinforcement Learning","date":"2020-04-17","arxiv_id":"2004.11145","repositories_listed":0,"syntology":null},{"url":null,"slug":"goal-conditioned-batch-reinforcement-learning","title":"Goal-conditioned Batch Reinforcement Learning for Rotation Invariant Locomotion","date":"2020-04-17","arxiv_id":"2004.08356","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-guided-deep-reinforcement-learning","title":"Knowledge-guided Deep Reinforcement Learning for Interactive Recommendation","date":"2020-04-17","arxiv_id":"2004.08068","repositories_listed":0,"syntology":null},{"url":null,"slug":"show-us-the-way-learning-to-manage-dialog","title":"Show Us the Way: Learning to Manage Dialog from Demonstrations","date":"2020-04-17","arxiv_id":"2004.08114","repositories_listed":0,"syntology":null},{"url":"/paper/a-game-theoretic-framework-for-model-based","slug":"a-game-theoretic-framework-for-model-based","title":"A Game Theoretic Framework for Model Based Reinforcement Learning","date":"2020-04-16","arxiv_id":"2004.07804","repositories_listed":0,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-game-theoretic-framework-for-model-based#ran","syntology_url":"https://syntology.ai/paper/2004.07804","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.07804"}},"official":null}},{"url":null,"slug":"data-driven-robust-control-using","title":"Data-Driven Robust Control Using Reinforcement Learning","date":"2020-04-16","arxiv_id":"2004.07690","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-safety-critical","title":"Reinforcement Learning for Safety-Critical Control under Model Uncertainty, using Control Lyapunov Functions and Control Barrier Functions","date":"2020-04-16","arxiv_id":"2004.07584","repositories_listed":0,"syntology":null},{"url":null,"slug":"actionspotter-deep-reinforcement-learning","title":"ActionSpotter: Deep Reinforcement Learning Framework for Temporal Action Spotting in Videos","date":"2020-04-15","arxiv_id":"2004.06971","repositories_listed":0,"syntology":null},{"url":null,"slug":"extending-deep-reinforcement-learning","title":"Extending Deep Reinforcement Learning Frameworks in Cryptocurrency Market Making","date":"2020-04-15","arxiv_id":"2004.06985","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-input-output-linearizing","title":"Improving Input-Output Linearizing Controllers for Bipedal Robots via Reinforcement Learning","date":"2020-04-15","arxiv_id":"2004.07276","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-deep-reinforcement-learning-based","title":"Safe deep reinforcement learning-based constrained optimal control scheme for active distribution networks","date":"2020-04-15","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-demonstration-of-issues-with-value-based","title":"A Demonstration of Issues with Value-Based Multiobjective Reinforcement Learning Under Stochastic State Transitions","date":"2020-04-14","arxiv_id":"2004.06277","repositories_listed":0,"syntology":null}],"record_sha256":"6485181722949250bef1fe6730333fa71cc13d1d5ab645fade58dea056896636","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}