{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/91","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":91,"pages_in_order":132,"rows_per_page":100,"rows":[9001,9100],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/90","next":"/task/reinforcement-learning/papers/92","papers":[{"url":null,"slug":"jointly-trained-state-action-embedding-for","title":"Jointly-Trained State-Action Embedding for Efficient Reinforcement Learning","date":"2020-09-28","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"neuron-activation-analysis-for-multi-joint","title":"Neuron Activation Analysis for Multi-Joint Robot Reinforcement Learning","date":"2020-09-28","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"repaint-knowledge-transfer-in-deep-actor","title":"REPAINT: Knowledge Transfer in Deep Actor-Critic Reinforcement Learning","date":"2020-09-28","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-emergence-of-individuality-in-multi-agent-1","title":"The Emergence of Individuality in Multi-Agent Reinforcement Learning","date":"2020-09-28","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-heterogeneous-multi-agent","title":"Towards Heterogeneous Multi-Agent Reinforcement Learning with Graph Neural Networks","date":"2020-09-28","arxiv_id":"2009.13161","repositories_listed":0,"syntology":null},{"url":null,"slug":"machine-learning-in-event-triggered-control","title":"Machine Learning in Event-Triggered Control: Recent Advances and Open Issues","date":"2020-09-27","arxiv_id":"2009.12783","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-deep-reinforcement-learning-for-ride","title":"Scalable Deep Reinforcement Learning for Ride-Hailing","date":"2020-09-27","arxiv_id":"2009.14679","repositories_listed":0,"syntology":null},{"url":null,"slug":"virtual-experience-to-real-world-application","title":"Virtual Experience to Real World Application: Sidewalk Obstacle Avoidance Using Reinforcement Learning for Visually Impaired","date":"2020-09-27","arxiv_id":"2009.12877","repositories_listed":0,"syntology":null},{"url":null,"slug":"complementary-meta-reinforcement-learning-for","title":"Complementary Meta-Reinforcement Learning for Fault-Adaptive Control","date":"2020-09-26","arxiv_id":"2009.12634","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-neural-induction-of-value-iteration","title":"Graph neural induction of value iteration","date":"2020-09-26","arxiv_id":"2009.12604","repositories_listed":0,"syntology":null},{"url":null,"slug":"lineage-evolution-reinforcement-learning","title":"Lineage Evolution Reinforcement Learning","date":"2020-09-26","arxiv_id":"2010.14616","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-n-ary-cross","title":"Reinforcement Learning-based N-ary Cross-Sentence Relation Extraction","date":"2020-09-26","arxiv_id":"2009.12683","repositories_listed":0,"syntology":null},{"url":null,"slug":"motion-planning-by-reinforcement-learning-for","title":"Motion Planning by Reinforcement Learning for an Unmanned Aerial Vehicle in Virtual Open Space with Static Obstacles","date":"2020-09-24","arxiv_id":"2009.11799","repositories_listed":0,"syntology":null},{"url":null,"slug":"sim-to-real-transfer-in-deep-reinforcement","title":"Sim-to-Real Transfer in Deep Reinforcement Learning for Robotics: a Survey","date":"2020-09-24","arxiv_id":"2009.13303","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-on-line","title":"Deep Reinforcement Learning for On-line Dialogue State Tracking","date":"2020-09-22","arxiv_id":"2009.10321","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-structured-actor-critic","title":"Distributed Structured Actor-Critic Reinforcement Learning for Universal Dialogue Management","date":"2020-09-22","arxiv_id":"2009.10326","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-horizon-value-estimation-for-model","title":"Dynamic Horizon Value Estimation for Model-based Reinforcement Learning","date":"2020-09-21","arxiv_id":"2009.09593","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-engagement-providing-evaluative-and","title":"Human Engagement Providing Evaluative and Informative Advice for Interactive Reinforcement Learning","date":"2020-09-21","arxiv_id":"2009.09575","repositories_listed":0,"syntology":null},{"url":null,"slug":"mobile-cellular-connected-uavs-reinforcement","title":"Mobile Cellular-Connected UAVs: Reinforcement Learning for Sky Limits","date":"2020-09-21","arxiv_id":"2009.09815","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-approaches-in-social","title":"Reinforcement Learning Approaches in Social Robotics","date":"2020-09-21","arxiv_id":"2009.09689","repositories_listed":0,"syntology":null},{"url":null,"slug":"lyapunov-based-reinforcement-learning-for","title":"Lyapunov-Based Reinforcement Learning for Decentralized Multi-Agent Control","date":"2020-09-20","arxiv_id":"2009.09361","repositories_listed":0,"syntology":null},{"url":null,"slug":"regret-bounds-and-reinforcement-learning","title":"Regret Bounds and Reinforcement Learning Exploration of EXP-based Algorithms","date":"2020-09-20","arxiv_id":"2009.09538","repositories_listed":0,"syntology":null},{"url":null,"slug":"construction-of-polar-codes-with","title":"Construction of Polar Codes with Reinforcement Learning","date":"2020-09-19","arxiv_id":"2009.09277","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-contraction-approach-to-model-based","title":"A Contraction Approach to Model-based Reinforcement Learning","date":"2020-09-18","arxiv_id":"2009.08586","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-closed-loop","title":"Deep Reinforcement Learning for Closed-Loop Blood Glucose Control","date":"2020-09-18","arxiv_id":"2009.09051","repositories_listed":0,"syntology":null},{"url":null,"slug":"private-reinforcement-learning-with-pac-and","title":"Private Reinforcement Learning with PAC and Regret Guarantees","date":"2020-09-18","arxiv_id":"2009.09052","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-relationship-between-dynamic-programming","title":"Reward Maximisation through Discrete Active Inference","date":"2020-09-17","arxiv_id":"2009.08111","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-behavior-level-explanation-for-deep","title":"Reconstructing Actions To Explain Deep Reinforcement Learning","date":"2020-09-17","arxiv_id":"2009.08507","repositories_listed":0,"syntology":null},{"url":null,"slug":"theory-of-mind-with-guilt-aversion","title":"Theory of Mind with Guilt Aversion Facilitates Cooperative Reinforcement Learning","date":"2020-09-16","arxiv_id":"2009.07445","repositories_listed":0,"syntology":null},{"url":null,"slug":"time-your-hedge-with-deep-reinforcement","title":"Time your hedge with Deep Reinforcement Learning","date":"2020-09-16","arxiv_id":"2009.14136","repositories_listed":0,"syntology":null},{"url":null,"slug":"transfer-learning-in-deep-reinforcement","title":"Transfer Learning in Deep Reinforcement Learning: A Survey","date":"2020-09-16","arxiv_id":"2009.07888","repositories_listed":0,"syntology":null},{"url":null,"slug":"decoding-polar-codes-with-reinforcement","title":"Decoding Polar Codes with Reinforcement Learning","date":"2020-09-15","arxiv_id":"2009.06796","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-strategic","title":"Reinforcement Learning for Strategic Recommendations","date":"2020-09-15","arxiv_id":"2009.07346","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-in-cournot","title":"Multi-Agent Reinforcement Learning in Cournot Games","date":"2020-09-14","arxiv_id":"2009.06224","repositories_listed":0,"syntology":null},{"url":null,"slug":"predictive-synthesis-of-quantum-materials-by","title":"Predictive Synthesis of Quantum Materials by Probabilistic Reinforcement Learning","date":"2020-09-14","arxiv_id":"2009.06739","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-dynamic-resource","title":"Reinforcement Learning for Dynamic Resource Optimization in 5G Radio Access Network Slicing","date":"2020-09-14","arxiv_id":"2009.06579","repositories_listed":0,"syntology":null},{"url":null,"slug":"extended-radial-basis-function-controller-for","title":"Extended Radial Basis Function Controller for Reinforcement Learning","date":"2020-09-12","arxiv_id":"2009.05866","repositories_listed":0,"syntology":null},{"url":null,"slug":"covid-19-pandemic-cyclic-lockdown","title":"COVID-19 Pandemic Cyclic Lockdown Optimization Using Reinforcement Learning","date":"2020-09-10","arxiv_id":"2009.04647","repositories_listed":0,"syntology":null},{"url":null,"slug":"importance-weighted-policy-learning-and","title":"Importance Weighted Policy Learning and Adaptation","date":"2020-09-10","arxiv_id":"2009.04875","repositories_listed":0,"syntology":null},{"url":null,"slug":"rlcfr-minimize-counterfactual-regret-by-deep","title":"RLCFR: Minimize Counterfactual Regret by Deep Reinforcement Learning","date":"2020-09-10","arxiv_id":"2009.06373","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-option","title":"Deep Reinforcement Learning for Option Replication and Hedging","date":"2020-09-09","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-objective-reinforcement-learning-for","title":"Multi-Objective Model-based Reinforcement Learning for Infectious Disease Control","date":"2020-09-09","arxiv_id":"2009.04607","repositories_listed":0,"syntology":null},{"url":null,"slug":"evolutionary-reinforcement-learning-via","title":"Evolutionary Reinforcement Learning via Cooperative Coevolutionary Negatively Correlated Search","date":"2020-09-08","arxiv_id":"2009.03603","repositories_listed":0,"syntology":null},{"url":null,"slug":"induction-and-exploitation-of-subgoal","title":"Induction and Exploitation of Subgoal Automata for Reinforcement Learning","date":"2020-09-08","arxiv_id":"2009.03855","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-learning-of-causal-structures-with","title":"Active Learning of Causal Structures with Deep Reinforcement Learning","date":"2020-09-07","arxiv_id":"2009.03009","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-learning-and-reinforcement-learning-for","title":"Deep Learning and Reinforcement Learning for Autonomous Unmanned Aerial Systems: Roadmap for Theory to Deployment","date":"2020-09-07","arxiv_id":"2009.03349","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-hybrid-pac-reinforcement-learning-algorithm","title":"A Hybrid PAC Reinforcement Learning Algorithm","date":"2020-09-05","arxiv_id":"2009.02602","repositories_listed":0,"syntology":null},{"url":null,"slug":"pac-reinforcement-learning-algorithm-for","title":"PAC Reinforcement Learning Algorithm for General-Sum Markov Games","date":"2020-09-05","arxiv_id":"2009.02605","repositories_listed":0,"syntology":null},{"url":null,"slug":"visualizing-the-loss-landscape-of-actor","title":"Visualizing the Loss Landscape of Actor Critic Methods with Applications in Inventory Optimization","date":"2020-09-04","arxiv_id":"2009.02391","repositories_listed":0,"syntology":null},{"url":null,"slug":"tap-net-transport-and-pack-using","title":"TAP-Net: Transport-and-Pack using Reinforcement Learning","date":"2020-09-03","arxiv_id":"2009.01469","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-approach-to-hybrid","title":"A reinforcement learning approach to hybrid control design","date":"2020-09-02","arxiv_id":"2009.00821","repositories_listed":0,"syntology":null},{"url":null,"slug":"plotthread-creating-expressive-storyline","title":"PlotThread: Creating Expressive Storyline Visualizations using Reinforcement Learning","date":"2020-09-01","arxiv_id":"2009.00249","repositories_listed":0,"syntology":null},{"url":null,"slug":"control-of-a-nature-inspired-scorpion-using","title":"Control of a Nature-inspired Scorpion using Reinforcement Learning","date":"2020-08-31","arxiv_id":"2008.13712","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-in-the-loop-methods-for-data-driven-and","title":"Human-in-the-Loop Methods for Data-Driven and Reinforcement Learning Systems","date":"2020-08-30","arxiv_id":"2008.13221","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-feedback","title":"Reinforcement Learning with Feedback-modulated TD-STDP","date":"2020-08-29","arxiv_id":"2008.13044","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-based-lane-change","title":"Meta Reinforcement Learning-Based Lane Change Strategy for Autonomous Vehicles","date":"2020-08-28","arxiv_id":"2008.12451","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-world-video-adaptation-with","title":"Real-world Video Adaptation with Reinforcement Learning","date":"2020-08-28","arxiv_id":"2008.12858","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-efficiency-in-sparse-reinforcement","title":"Sample Efficiency in Sparse Reinforcement Learning: Or Your Money Back","date":"2020-08-28","arxiv_id":"2008.12693","repositories_listed":0,"syntology":null},{"url":null,"slug":"document-editing-assistants-and-model-based","title":"Document-editing Assistants and Model-based Reinforcement Learning as a Path to Conversational AI","date":"2020-08-27","arxiv_id":"2008.12095","repositories_listed":0,"syntology":null},{"url":null,"slug":"synthetic-sample-selection-via-reinforcement","title":"Synthetic Sample Selection via Reinforcement Learning","date":"2020-08-26","arxiv_id":"2008.11331","repositories_listed":0,"syntology":null},{"url":null,"slug":"auxiliary-task-based-deep-reinforcement","title":"Auxiliary-task Based Deep Reinforcement Learning for Participant Selection Problem in Mobile Crowdsourcing","date":"2020-08-25","arxiv_id":"2008.11087","repositories_listed":0,"syntology":null},{"url":null,"slug":"ensuring-monotonic-policy-improvement-in","title":"Ensuring Monotonic Policy Improvement in Entropy-regularized Value-based Reinforcement Learning","date":"2020-08-25","arxiv_id":"2008.10806","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-reinforcement-learning-a-case-study-in","title":"Robust Reinforcement Learning: A Case Study in Linear Quadratic Regulation","date":"2020-08-25","arxiv_id":"2008.11592","repositories_listed":0,"syntology":null},{"url":null,"slug":"t-soft-update-of-target-network-for-deep","title":"t-Soft Update of Target Network for Deep Reinforcement Learning","date":"2020-08-25","arxiv_id":"2008.10861","repositories_listed":0,"syntology":null},{"url":null,"slug":"variable-compliance-control-for-robotic-peg","title":"Variable Compliance Control for Robotic Peg-in-Hole Assembly: A Deep Reinforcement Learning Approach","date":"2020-08-24","arxiv_id":"2008.10224","repositories_listed":0,"syntology":null},{"url":null,"slug":"dsp-a-differential-spatial-prediction-scheme","title":"DSP: A Differential Spatial Prediction Scheme for Comprehensive real industrial datasets","date":"2020-08-23","arxiv_id":"2008.09951","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-imitation-learning-via-random","title":"Adversarial Imitation Learning via Random Search","date":"2020-08-21","arxiv_id":"2008.09450","repositories_listed":0,"syntology":null},{"url":"/paper/model-free-episodic-control-with-state","slug":"model-free-episodic-control-with-state","title":"Model-Free Episodic Control with State Aggregation","date":"2020-08-21","arxiv_id":"2008.09685","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-admission","title":"Reinforcement Learning-based Admission Control in Delay-sensitive Service Systems","date":"2020-08-21","arxiv_id":"2008.09590","repositories_listed":0,"syntology":null},{"url":null,"slug":"static-neural-compiler-optimization-via-deep","title":"Static Neural Compiler Optimization via Deep Reinforcement Learning","date":"2020-08-20","arxiv_id":"2008.08951","repositories_listed":0,"syntology":null},{"url":null,"slug":"intelligent-replication-management-for-hdfs","title":"Intelligent Replication Management for HDFS Using Reinforcement Learning","date":"2020-08-19","arxiv_id":"2008.08665","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-framework-for-studying-reinforcement","title":"A Framework for Studying Reinforcement Learning and Sim-to-Real in Robot Soccer","date":"2020-08-18","arxiv_id":"2008.12624","repositories_listed":0,"syntology":null},{"url":null,"slug":"analysis-of-social-robotic-navigation","title":"Analysis of Social Robotic Navigation approaches: CNN Encoder and Incremental Learning as an alternative to Deep Reinforcement Learning","date":"2020-08-18","arxiv_id":"2008.07965","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-improving-object","title":"Reinforcement Learning for Improving Object Detection","date":"2020-08-18","arxiv_id":"2008.08005","repositories_listed":0,"syntology":null},{"url":null,"slug":"relmogen-leveraging-motion-generation-in","title":"ReLMoGen: Leveraging Motion Generation in Reinforcement Learning for Mobile Manipulation","date":"2020-08-18","arxiv_id":"2008.07792","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-reinforcement-learning-for","title":"A Survey on Reinforcement Learning for Combinatorial Optimization","date":"2020-08-17","arxiv_id":"2008.12248","repositories_listed":0,"syntology":null},{"url":null,"slug":"deepslicing-deep-reinforcement-learning","title":"DeepSlicing: Deep Reinforcement Learning Assisted Resource Allocation for Network Slicing","date":"2020-08-17","arxiv_id":"2008.07614","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-reference-reinforcement-learning-for","title":"Model-Reference Reinforcement Learning for Collision-Free Tracking Control of Autonomous Surface Vehicles","date":"2020-08-17","arxiv_id":"2008.07240","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-adaptive-synchronization-approach-for","title":"An adaptive synchronization approach for weights of deep reinforcement learning","date":"2020-08-16","arxiv_id":"2008.06973","repositories_listed":0,"syntology":null},{"url":null,"slug":"drl-based-qos-aware-resource-allocation","title":"DRL-Based QoS-Aware Resource Allocation Scheme for Coexistence of Licensed and Unlicensed Users in LTE and Beyond","date":"2020-08-16","arxiv_id":"2008.06905","repositories_listed":0,"syntology":null},{"url":null,"slug":"inverse-reinforcement-learning-with-natural","title":"Inverse Reinforcement Learning with Natural Language Goals","date":"2020-08-16","arxiv_id":"2008.06924","repositories_listed":0,"syntology":null},{"url":null,"slug":"chrome-dino-run-using-reinforcement-learning","title":"Chrome Dino Run using Reinforcement Learning","date":"2020-08-15","arxiv_id":"2008.06799","repositories_listed":0,"syntology":null},{"url":null,"slug":"explainability-in-deep-reinforcement-learning","title":"Explainability in Deep Reinforcement Learning","date":"2020-08-15","arxiv_id":"2008.06693","repositories_listed":0,"syntology":null},{"url":null,"slug":"defending-adversarial-attacks-without","title":"Adversary Agnostic Robust Deep Reinforcement Learning","date":"2020-08-14","arxiv_id":"2008.06199","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-trajectory","title":"Reinforcement Learning with Trajectory Feedback","date":"2020-08-13","arxiv_id":"2008.06036","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-ocular-biomechanics-environment-for","title":"An ocular biomechanics environment for reinforcement learning","date":"2020-08-12","arxiv_id":"2008.05088","repositories_listed":0,"syntology":null},{"url":null,"slug":"overcoming-model-bias-for-robust-offline-deep","title":"Overcoming Model Bias for Robust Offline Deep Reinforcement Learning","date":"2020-08-12","arxiv_id":"2008.05533","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-intelligent-control-strategy-for-buck-dc","title":"An Intelligent Control Strategy for buck DC-DC Converter via Deep Reinforcement Learning","date":"2020-08-11","arxiv_id":"2008.04542","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-deep-reinforcement-learning-for-1","title":"Deep Model-Based Reinforcement Learning for High-Dimensional Problems, a Survey","date":"2020-08-11","arxiv_id":"2008.05598","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparison-of-model-predictive-and","title":"Comparison of Model Predictive and Reinforcement Learning Methods for Fault Tolerant Control","date":"2020-08-10","arxiv_id":"2008.04403","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-deep-reinforcement-learning-for-1","title":"Distributed Deep Reinforcement Learning for Functional Split Control in Energy Harvesting Virtualized Small Cells","date":"2020-08-07","arxiv_id":"2008.04105","repositories_listed":0,"syntology":null},{"url":null,"slug":"managing-caching-strategies-for-stream","title":"Managing caching strategies for stream reasoning with reinforcement learning","date":"2020-08-07","arxiv_id":"2008.03212","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-gentle-lecture-note-on-filtrations-in","title":"A Gentle Lecture Note on Filtrations in Reinforcement Learning","date":"2020-08-06","arxiv_id":"2008.02622","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-to-detect-brain","title":"Deep reinforcement learning to detect brain lesions on MRI: a proof-of-concept application of reinforcement learning to medical images","date":"2020-08-06","arxiv_id":"2008.02708","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-field","title":"Deep Reinforcement Learning for Field Development Optimization","date":"2020-08-05","arxiv_id":"2008.12627","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-ad-pruning-of-sponsored-search","title":"Optimizing AD Pruning of Sponsored Search with Reinforcement Learning","date":"2020-08-05","arxiv_id":"2008.02014","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-driven-information","title":"Reinforcement Learning-driven Information Seeking: A Quantum Probabilistic Approach","date":"2020-08-05","arxiv_id":"2008.02372","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparative-analysis-of-deep-reinforcement","title":"A Comparative Analysis of Deep Reinforcement Learning-enabled Freeway Decision-making for Automated Vehicles","date":"2020-08-04","arxiv_id":"2008.01302","repositories_listed":0,"syntology":null},{"url":null,"slug":"easyrl-a-simple-and-extensible-reinforcement","title":"EasyRL: A Simple and Extensible Reinforcement Learning Framework","date":"2020-08-04","arxiv_id":"2008.01700","repositories_listed":0,"syntology":null},{"url":null,"slug":"explanation-of-reinforcement-learning-model","title":"Explanation of Reinforcement Learning Model in Dynamic Multi-Agent System","date":"2020-08-04","arxiv_id":"2008.01508","repositories_listed":0,"syntology":null}],"record_sha256":"a274c3dd042580a0cef4d00b12ff6484b52c2a1c3bbe03cca16f4428b8c0c9a6","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}