{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/105","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":105,"pages_in_order":135,"rows_per_page":100,"rows":[10401,10500],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/104","next":"/task/reinforcement-learning-2/papers/106","papers":[{"url":null,"slug":"sim-to-real-transfer-in-deep-reinforcement","title":"Sim-to-Real Transfer in Deep Reinforcement Learning for Robotics: a Survey","date":"2020-09-24","arxiv_id":"2009.13303","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multi-agent-deep-reinforcement-learning","title":"A Multi-Agent Deep Reinforcement Learning Approach for a Distributed Energy Marketplace in Smart Grids","date":"2020-09-23","arxiv_id":"2009.10905","repositories_listed":0,"syntology":null},{"url":null,"slug":"demand-responsive-dynamic-pricing-framework","title":"Demand Responsive Dynamic Pricing Framework for Prosumer Dominated Microgrids using Multiagent Reinforcement Learning","date":"2020-09-23","arxiv_id":"2009.10890","repositories_listed":0,"syntology":null},{"url":null,"slug":"probabilistic-machine-learning-for-healthcare","title":"Probabilistic Machine Learning for Healthcare","date":"2020-09-23","arxiv_id":"2009.11087","repositories_listed":0,"syntology":null},{"url":null,"slug":"releaser-a-reinforcement-learning-strategy","title":"ReLeaSER: A Reinforcement Learning Strategy for Optimizing Utilization Of Ephemeral Cloud Resources","date":"2020-09-23","arxiv_id":"2009.11208","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-reinforcement-learning-based","title":"Robust Reinforcement Learning-based Autonomous Driving Agent for Simulation and Real World","date":"2020-09-23","arxiv_id":"2009.11212","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-on-line","title":"Deep Reinforcement Learning for On-line Dialogue State Tracking","date":"2020-09-22","arxiv_id":"2009.10321","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-structured-actor-critic","title":"Distributed Structured Actor-Critic Reinforcement Learning for Universal Dialogue Management","date":"2020-09-22","arxiv_id":"2009.10326","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-q-learning-provably-efficient-an-extended","title":"Is Q-Learning Provably Efficient? An Extended Analysis","date":"2020-09-22","arxiv_id":"2009.10396","repositories_listed":0,"syntology":null},{"url":null,"slug":"sumbt-larl-end-to-end-neural-task-oriented","title":"SUMBT+LaRL: Effective Multi-domain End-to-end Neural Task-oriented Dialog System","date":"2020-09-22","arxiv_id":"2009.10447","repositories_listed":0,"syntology":null},{"url":null,"slug":"dispatch-design-space-exploration-of-cyber","title":"DISPATCH: Design Space Exploration of Cyber-Physical Systems","date":"2020-09-21","arxiv_id":"2009.10214","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-horizon-value-estimation-for-model","title":"Dynamic Horizon Value Estimation for Model-based Reinforcement Learning","date":"2020-09-21","arxiv_id":"2009.09593","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-engagement-providing-evaluative-and","title":"Human Engagement Providing Evaluative and Informative Advice for Interactive Reinforcement Learning","date":"2020-09-21","arxiv_id":"2009.09575","repositories_listed":0,"syntology":null},{"url":null,"slug":"learn-to-exceed-stereo-inverse-reinforcement","title":"Learn to Exceed: Stereo Inverse Reinforcement Learning with Concurrent Policy Optimization","date":"2020-09-21","arxiv_id":"2009.09577","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-a-contact-adaptive-controller-for","title":"Learning a Contact-Adaptive Controller for Robust, Efficient Legged Locomotion","date":"2020-09-21","arxiv_id":"2009.10019","repositories_listed":0,"syntology":null},{"url":null,"slug":"mobile-cellular-connected-uavs-reinforcement","title":"Mobile Cellular-Connected UAVs: Reinforcement Learning for Sky Limits","date":"2020-09-21","arxiv_id":"2009.09815","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-approaches-in-social","title":"Reinforcement Learning Approaches in Social Robotics","date":"2020-09-21","arxiv_id":"2009.09689","repositories_listed":0,"syntology":null},{"url":null,"slug":"lyapunov-based-reinforcement-learning-for","title":"Lyapunov-Based Reinforcement Learning for Decentralized Multi-Agent Control","date":"2020-09-20","arxiv_id":"2009.09361","repositories_listed":0,"syntology":null},{"url":null,"slug":"multiplayer-support-for-the-arcade-learning","title":"Multiplayer Support for the Arcade Learning Environment","date":"2020-09-20","arxiv_id":"2009.09341","repositories_listed":0,"syntology":null},{"url":null,"slug":"regret-bounds-and-reinforcement-learning","title":"Regret Bounds and Reinforcement Learning Exploration of EXP-based Algorithms","date":"2020-09-20","arxiv_id":"2009.09538","repositories_listed":0,"syntology":null},{"url":null,"slug":"construction-of-polar-codes-with","title":"Construction of Polar Codes with Reinforcement Learning","date":"2020-09-19","arxiv_id":"2009.09277","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-contraction-approach-to-model-based","title":"A Contraction Approach to Model-based Reinforcement Learning","date":"2020-09-18","arxiv_id":"2009.08586","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-closed-loop","title":"Deep Reinforcement Learning for Closed-Loop Blood Glucose Control","date":"2020-09-18","arxiv_id":"2009.09051","repositories_listed":0,"syntology":null},{"url":null,"slug":"private-reinforcement-learning-with-pac-and","title":"Private Reinforcement Learning with PAC and Regret Guarantees","date":"2020-09-18","arxiv_id":"2009.09052","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-weakly-supervised","title":"Reinforcement Learning for Weakly Supervised Temporal Grounding of Natural Language in Untrimmed Videos","date":"2020-09-18","arxiv_id":"2009.08614","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalight-improving-environment","title":"GeneraLight: Improving Environment Generalization of Traffic Signal Control via Meta Reinforcement Learning","date":"2020-09-17","arxiv_id":"2009.08052","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-relationship-between-dynamic-programming","title":"Reward Maximisation through Discrete Active Inference","date":"2020-09-17","arxiv_id":"2009.08111","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-behavior-level-explanation-for-deep","title":"Reconstructing Actions To Explain Deep Reinforcement Learning","date":"2020-09-17","arxiv_id":"2009.08507","repositories_listed":0,"syntology":null},{"url":null,"slug":"theory-of-mind-with-guilt-aversion","title":"Theory of Mind with Guilt Aversion Facilitates Cooperative Reinforcement Learning","date":"2020-09-16","arxiv_id":"2009.07445","repositories_listed":0,"syntology":null},{"url":null,"slug":"time-your-hedge-with-deep-reinforcement","title":"Time your hedge with Deep Reinforcement Learning","date":"2020-09-16","arxiv_id":"2009.14136","repositories_listed":0,"syntology":null},{"url":null,"slug":"transfer-learning-in-deep-reinforcement","title":"Transfer Learning in Deep Reinforcement Learning: A Survey","date":"2020-09-16","arxiv_id":"2009.07888","repositories_listed":0,"syntology":null},{"url":null,"slug":"decoding-polar-codes-with-reinforcement","title":"Decoding Polar Codes with Reinforcement Learning","date":"2020-09-15","arxiv_id":"2009.06796","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-strategic","title":"Reinforcement Learning for Strategic Recommendations","date":"2020-09-15","arxiv_id":"2009.07346","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-transformers-a-survey","title":"Efficient Transformers: A Survey","date":"2020-09-14","arxiv_id":"2009.06732","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-in-cournot","title":"Multi-Agent Reinforcement Learning in Cournot Games","date":"2020-09-14","arxiv_id":"2009.06224","repositories_listed":0,"syntology":null},{"url":null,"slug":"predictive-synthesis-of-quantum-materials-by","title":"Predictive Synthesis of Quantum Materials by Probabilistic Reinforcement Learning","date":"2020-09-14","arxiv_id":"2009.06739","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-dynamic-resource","title":"Reinforcement Learning for Dynamic Resource Optimization in 5G Radio Access Network Slicing","date":"2020-09-14","arxiv_id":"2009.06579","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-competitive-self-play-policy","title":"Efficient Competitive Self-Play Policy Optimization","date":"2020-09-13","arxiv_id":"2009.06086","repositories_listed":0,"syntology":null},{"url":null,"slug":"extended-radial-basis-function-controller-for","title":"Extended Radial Basis Function Controller for Reinforcement Learning","date":"2020-09-12","arxiv_id":"2009.05866","repositories_listed":0,"syntology":null},{"url":null,"slug":"guided-policy-search-based-control-of-a-high","title":"Guided Policy Search Based Control of a High Dimensional Advanced Manufacturing Process","date":"2020-09-12","arxiv_id":"2009.05838","repositories_listed":0,"syntology":null},{"url":null,"slug":"embodied-visual-navigation-with-automatic","title":"Embodied Visual Navigation with Automatic Curriculum Learning in Real Environments","date":"2020-09-11","arxiv_id":"2009.05429","repositories_listed":0,"syntology":null},{"url":null,"slug":"covid-19-pandemic-cyclic-lockdown","title":"COVID-19 Pandemic Cyclic Lockdown Optimization Using Reinforcement Learning","date":"2020-09-10","arxiv_id":"2009.04647","repositories_listed":0,"syntology":null},{"url":null,"slug":"importance-weighted-policy-learning-and","title":"Importance Weighted Policy Learning and Adaptation","date":"2020-09-10","arxiv_id":"2009.04875","repositories_listed":0,"syntology":null},{"url":null,"slug":"rlcfr-minimize-counterfactual-regret-by-deep","title":"RLCFR: Minimize Counterfactual Regret by Deep Reinforcement Learning","date":"2020-09-10","arxiv_id":"2009.06373","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-option","title":"Deep Reinforcement Learning for Option Replication and Hedging","date":"2020-09-09","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-objective-reinforcement-learning-for","title":"Multi-Objective Model-based Reinforcement Learning for Infectious Disease Control","date":"2020-09-09","arxiv_id":"2009.04607","repositories_listed":0,"syntology":null},{"url":null,"slug":"qr-mix-distributional-value-function","title":"QR-MIX: Distributional Value Function Factorisation for Cooperative Multi-Agent Reinforcement Learning","date":"2020-09-09","arxiv_id":"2009.04197","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-non-stationary-1","title":"Reinforcement Learning in Non-Stationary Discrete-Time Linear-Quadratic Mean-Field Games","date":"2020-09-09","arxiv_id":"2009.04350","repositories_listed":0,"syntology":null},{"url":null,"slug":"evolutionary-reinforcement-learning-via","title":"Evolutionary Reinforcement Learning via Cooperative Coevolutionary Negatively Correlated Search","date":"2020-09-08","arxiv_id":"2009.03603","repositories_listed":0,"syntology":null},{"url":null,"slug":"induction-and-exploitation-of-subgoal","title":"Induction and Exploitation of Subgoal Automata for Reinforcement Learning","date":"2020-09-08","arxiv_id":"2009.03855","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-on-job-shop-scheduling","title":"Graph neural networks-based Scheduler for Production planning problems using Reinforcement Learning","date":"2020-09-08","arxiv_id":"2009.03836","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-learning-of-causal-structures-with","title":"Active Learning of Causal Structures with Deep Reinforcement Learning","date":"2020-09-07","arxiv_id":"2009.03009","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-learning-and-reinforcement-learning-for","title":"Deep Learning and Reinforcement Learning for Autonomous Unmanned Aerial Systems: Roadmap for Theory to Deployment","date":"2020-09-07","arxiv_id":"2009.03349","repositories_listed":0,"syntology":null},{"url":null,"slug":"detecting-and-adapting-to-crisis-pattern-with","title":"Detecting and adapting to crisis pattern with context based Deep Reinforcement Learning","date":"2020-09-07","arxiv_id":"2009.07200","repositories_listed":0,"syntology":null},{"url":null,"slug":"driving-tasks-transfer-in-deep-reinforcement","title":"Driving Tasks Transfer in Deep Reinforcement Learning for Decision-making of Autonomous Vehicles","date":"2020-09-07","arxiv_id":"2009.03268","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-hybrid-pac-reinforcement-learning-algorithm","title":"A Hybrid PAC Reinforcement Learning Algorithm","date":"2020-09-05","arxiv_id":"2009.02602","repositories_listed":0,"syntology":null},{"url":null,"slug":"pac-reinforcement-learning-algorithm-for","title":"PAC Reinforcement Learning Algorithm for General-Sum Markov Games","date":"2020-09-05","arxiv_id":"2009.02605","repositories_listed":0,"syntology":null},{"url":null,"slug":"visualizing-the-loss-landscape-of-actor","title":"Visualizing the Loss Landscape of Actor Critic Methods with Applications in Inventory Optimization","date":"2020-09-04","arxiv_id":"2009.02391","repositories_listed":0,"syntology":null},{"url":null,"slug":"tap-net-transport-and-pack-using","title":"TAP-Net: Transport-and-Pack using Reinforcement Learning","date":"2020-09-03","arxiv_id":"2009.01469","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-approach-to-hybrid","title":"A reinforcement learning approach to hybrid control design","date":"2020-09-02","arxiv_id":"2009.00821","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-reinforcement-learning-model-for","title":"Adaptive Reinforcement Learning Model for Simulation of Urban Mobility during Crises","date":"2020-09-02","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"plotthread-creating-expressive-storyline","title":"PlotThread: Creating Expressive Storyline Visualizations using Reinforcement Learning","date":"2020-09-01","arxiv_id":"2009.00249","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-black-box","title":"Reinforcement Learning-based Black-Box Evasion Attacks to Link Prediction in Dynamic Graphs","date":"2020-09-01","arxiv_id":"2009.00163","repositories_listed":0,"syntology":null},{"url":null,"slug":"solving-the-single-track-train-scheduling","title":"Solving the single-track train scheduling problem via Deep Reinforcement Learning","date":"2020-09-01","arxiv_id":"2009.00433","repositories_listed":0,"syntology":null},{"url":null,"slug":"control-of-a-nature-inspired-scorpion-using","title":"Control of a Nature-inspired Scorpion using Reinforcement Learning","date":"2020-08-31","arxiv_id":"2008.13712","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-driven-outer-loop-control-using-deep","title":"Data-driven Outer-Loop Control Using Deep Reinforcement Learning for Trajectory Tracking","date":"2020-08-31","arxiv_id":"2008.13732","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-reinforcement-learning-in-factored","title":"Efficient Reinforcement Learning in Factored MDPs with Application to Constrained RL","date":"2020-08-31","arxiv_id":"2008.13319","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-contact-rich","title":"Deep Reinforcement Learning for Contact-Rich Skills Using Compliant Movement Primitives","date":"2020-08-30","arxiv_id":"2008.13223","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-in-the-loop-methods-for-data-driven-and","title":"Human-in-the-Loop Methods for Data-Driven and Reinforcement Learning Systems","date":"2020-08-30","arxiv_id":"2008.13221","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-feedback","title":"Reinforcement Learning with Feedback-modulated TD-STDP","date":"2020-08-29","arxiv_id":"2008.13044","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-based-lane-change","title":"Meta Reinforcement Learning-Based Lane Change Strategy for Autonomous Vehicles","date":"2020-08-28","arxiv_id":"2008.12451","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-world-video-adaptation-with","title":"Real-world Video Adaptation with Reinforcement Learning","date":"2020-08-28","arxiv_id":"2008.12858","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-efficiency-in-sparse-reinforcement","title":"Sample Efficiency in Sparse Reinforcement Learning: Or Your Money Back","date":"2020-08-28","arxiv_id":"2008.12693","repositories_listed":0,"syntology":null},{"url":null,"slug":"controlling-level-of-unconsciousness-with","title":"Controlling Level of Unconsciousness by Titrating Propofol with Deep Reinforcement Learning","date":"2020-08-27","arxiv_id":"2008.12333","repositories_listed":0,"syntology":null},{"url":null,"slug":"document-editing-assistants-and-model-based","title":"Document-editing Assistants and Model-based Reinforcement Learning as a Path to Conversational AI","date":"2020-08-27","arxiv_id":"2008.12095","repositories_listed":0,"syntology":null},{"url":null,"slug":"selective-particle-attention-visual-feature","title":"Selective Particle Attention: Visual Feature-Based Attention in Deep Reinforcement Learning","date":"2020-08-26","arxiv_id":"2008.11491","repositories_listed":0,"syntology":null},{"url":null,"slug":"synthetic-sample-selection-via-reinforcement","title":"Synthetic Sample Selection via Reinforcement Learning","date":"2020-08-26","arxiv_id":"2008.11331","repositories_listed":0,"syntology":null},{"url":null,"slug":"auxiliary-task-based-deep-reinforcement","title":"Auxiliary-task Based Deep Reinforcement Learning for Participant Selection Problem in Mobile Crowdsourcing","date":"2020-08-25","arxiv_id":"2008.11087","repositories_listed":0,"syntology":null},{"url":null,"slug":"ensuring-monotonic-policy-improvement-in","title":"Ensuring Monotonic Policy Improvement in Entropy-regularized Value-based Reinforcement Learning","date":"2020-08-25","arxiv_id":"2008.10806","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-reinforcement-learning-a-case-study-in","title":"Robust Reinforcement Learning: A Case Study in Linear Quadratic Regulation","date":"2020-08-25","arxiv_id":"2008.11592","repositories_listed":0,"syntology":null},{"url":null,"slug":"t-soft-update-of-target-network-for-deep","title":"t-Soft Update of Target Network for Deep Reinforcement Learning","date":"2020-08-25","arxiv_id":"2008.10861","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-dispatching-for-large-scale","title":"Dynamic Dispatching for Large-Scale Heterogeneous Fleet via Multi-agent Deep Reinforcement Learning","date":"2020-08-24","arxiv_id":"2008.10713","repositories_listed":0,"syntology":null},{"url":null,"slug":"variable-compliance-control-for-robotic-peg","title":"Variable Compliance Control for Robotic Peg-in-Hole Assembly: A Deep Reinforcement Learning Approach","date":"2020-08-24","arxiv_id":"2008.10224","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-and-multiple-time-scale-eligibility","title":"Adaptive and Multiple Time-scale Eligibility Traces for Online Deep Reinforcement Learning","date":"2020-08-23","arxiv_id":"2008.10040","repositories_listed":0,"syntology":null},{"url":null,"slug":"dsp-a-differential-spatial-prediction-scheme","title":"DSP: A Differential Spatial Prediction Scheme for Comprehensive real industrial datasets","date":"2020-08-23","arxiv_id":"2008.09951","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-imitation-learning-via-random","title":"Adversarial Imitation Learning via Random Search","date":"2020-08-21","arxiv_id":"2008.09450","repositories_listed":0,"syntology":null},{"url":null,"slug":"biomechanic-posture-stabilisation-via","title":"Biomechanic Posture Stabilisation via Iterative Training of Multi-policy Deep Reinforcement Learning Agents","date":"2020-08-21","arxiv_id":"2008.12210","repositories_listed":0,"syntology":null},{"url":"/paper/model-free-episodic-control-with-state","slug":"model-free-episodic-control-with-state","title":"Model-Free Episodic Control with State Aggregation","date":"2020-08-21","arxiv_id":"2008.09685","repositories_listed":0,"syntology":null},{"url":null,"slug":"nancy-neural-adaptive-network-coding","title":"NANCY: Neural Adaptive Network Coding methodologY for video distribution over wireless networks","date":"2020-08-21","arxiv_id":"2008.09559","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-admission","title":"Reinforcement Learning-based Admission Control in Delay-sensitive Service Systems","date":"2020-08-21","arxiv_id":"2008.09590","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-dynamic-weighing","title":"Reinforcement Learning based dynamic weighing of Ensemble Models for Time Series Forecasting","date":"2020-08-20","arxiv_id":"2008.08878","repositories_listed":0,"syntology":null},{"url":null,"slug":"static-neural-compiler-optimization-via-deep","title":"Static Neural Compiler Optimization via Deep Reinforcement Learning","date":"2020-08-20","arxiv_id":"2008.08951","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-knowledge-based-sequential","title":"A Survey of Knowledge-based Sequential Decision Making under Uncertainty","date":"2020-08-19","arxiv_id":"2008.08548","repositories_listed":0,"syntology":null},{"url":null,"slug":"intelligent-replication-management-for-hdfs","title":"Intelligent Replication Management for HDFS Using Reinforcement Learning","date":"2020-08-19","arxiv_id":"2008.08665","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-framework-for-studying-reinforcement","title":"A Framework for Studying Reinforcement Learning and Sim-to-Real in Robot Soccer","date":"2020-08-18","arxiv_id":"2008.12624","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-trading-strategies-across-liquidity","title":"Adaptive trading strategies across liquidity pools","date":"2020-08-18","arxiv_id":"2008.07807","repositories_listed":0,"syntology":null},{"url":null,"slug":"analysis-of-social-robotic-navigation","title":"Analysis of Social Robotic Navigation approaches: CNN Encoder and Incremental Learning as an alternative to Deep Reinforcement Learning","date":"2020-08-18","arxiv_id":"2008.07965","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-improving-object","title":"Reinforcement Learning for Improving Object Detection","date":"2020-08-18","arxiv_id":"2008.08005","repositories_listed":0,"syntology":null},{"url":null,"slug":"relmogen-leveraging-motion-generation-in","title":"ReLMoGen: Leveraging Motion Generation in Reinforcement Learning for Mobile Manipulation","date":"2020-08-18","arxiv_id":"2008.07792","repositories_listed":0,"syntology":null},{"url":null,"slug":"super-human-performance-in-gran-turismo-sport","title":"Super-Human Performance in Gran Turismo Sport Using Deep Reinforcement Learning","date":"2020-08-18","arxiv_id":"2008.07971","repositories_listed":0,"syntology":null}],"record_sha256":"30ed227a74264a174e1c0f5043e46b5f322df66ee7da71cad2eef672f9d8dafa","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}