{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/116","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":116,"pages_in_order":152,"rows_per_page":100,"rows":[11501,11600],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/115","next":"/task/reinforcement-learning-1/papers/117","papers":[{"url":null,"slug":"sentiment-analysis-for-reinforcement-learning","title":"Sentiment Analysis for Reinforcement Learning","date":"2020-10-05","arxiv_id":"2010.02316","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-act-of-remembering-a-study-in-partially-1","title":"The act of remembering: a study in partially observable reinforcement learning","date":"2020-10-05","arxiv_id":"2010.01753","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-sharp-analysis-of-model-based-reinforcement-1","title":"A Sharp Analysis of Model-based Reinforcement Learning with Self-Play","date":"2020-10-04","arxiv_id":"2010.01604","repositories_listed":0,"syntology":null},{"url":null,"slug":"test-cost-sensitive-methods-for-identifying","title":"Test-Cost Sensitive Methods for Identifying Nearby Points","date":"2020-10-04","arxiv_id":"2010.03962","repositories_listed":0,"syntology":null},{"url":null,"slug":"attractor-selection-in-nonlinear-energy","title":"Attractor Selection in Nonlinear Energy Harvesting Using Deep Reinforcement Learning","date":"2020-10-03","arxiv_id":"2010.01255","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-tabula-rasa-a-modular-reinforcement","title":"Beyond Tabula-Rasa: a Modular Reinforcement Learning Approach for Physically Embedded 3D Sokoban","date":"2020-10-03","arxiv_id":"2010.01298","repositories_listed":0,"syntology":null},{"url":null,"slug":"disentangling-causal-effects-for-hierarchical","title":"Disentangling causal effects for hierarchical reinforcement learning","date":"2020-10-03","arxiv_id":"2010.01351","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-gradient-with-expected-quadratic-1","title":"Mean-Variance Efficient Reinforcement Learning with Applications to Dynamic Financial Investment","date":"2020-10-03","arxiv_id":"2010.01404","repositories_listed":0,"syntology":null},{"url":null,"slug":"interactive-reinforcement-learning-for","title":"Interactive Reinforcement Learning for Feature Selection with Decision Tree in the Loop","date":"2020-10-02","arxiv_id":"2010.02506","repositories_listed":0,"syntology":null},{"url":null,"slug":"madras-multi-agent-driving-simulator","title":"MADRaS : Multi Agent Driving Simulator","date":"2020-10-02","arxiv_id":"2010.00993","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-of-simple-indirect","title":"Reinforcement Learning of Sequential Price Mechanisms","date":"2020-10-02","arxiv_id":"2010.01180","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-mixed","title":"Deep Reinforcement Learning with Mixed Convolutional Network","date":"2020-10-01","arxiv_id":"2010.00717","repositories_listed":0,"syntology":null},{"url":null,"slug":"minimax-optimal-reinforcement-learning-for","title":"Nearly Minimax Optimal Reinforcement Learning for Discounted MDPs","date":"2020-10-01","arxiv_id":"2010.00587","repositories_listed":0,"syntology":null},{"url":"/paper/multi-agent-social-reinforcement-learning","slug":"multi-agent-social-reinforcement-learning","title":"Emergent Social Learning via Multi-agent Reinforcement Learning","date":"2020-10-01","arxiv_id":"2010.00581","repositories_listed":0,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multi-agent-social-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2010.00581","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.00581"}},"official":null}},{"url":null,"slug":"multi-reward-based-reinforcement-learning-for","title":"Multi-Reward based Reinforcement Learning for Neural Machine Translation","date":"2020-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"recognition-method-of-important-words-in","title":"Recognition Method of Important Words in Korean Text based on Reinforcement Learning","date":"2020-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"value-based-bayesian-meta-reinforcement","title":"Bayesian Meta-reinforcement Learning for Traffic Signal Control","date":"2020-10-01","arxiv_id":"2010.00163","repositories_listed":0,"syntology":null},{"url":null,"slug":"aamdrl-augmented-asset-management-with-deep","title":"AAMDRL: Augmented Asset Management with Deep Reinforcement Learning","date":"2020-09-30","arxiv_id":"2010.08497","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-optimization-and-reinforcement","title":"Accelerating Optimization and Reinforcement Learning with Quasi-Stochastic Approximation","date":"2020-09-30","arxiv_id":"2009.14431","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-the-gap-between-markowitz-planning","title":"Bridging the gap between Markowitz planning and deep reinforcement learning","date":"2020-09-30","arxiv_id":"2010.09108","repositories_listed":0,"syntology":null},{"url":null,"slug":"entropy-regularization-for-mean-field-games","title":"Entropy Regularization for Mean Field Games with Learning","date":"2020-09-30","arxiv_id":"2010.00145","repositories_listed":0,"syntology":null},{"url":null,"slug":"finding-it-at-another-side-a-viewpoint-1","title":"Finding It at Another Side: A Viewpoint-Adapted Matching Encoder for Change Captioning","date":"2020-09-30","arxiv_id":"2009.14352","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-based-heuristic-search-for-module","title":"Graph-based Heuristic Search for Module Selection Procedure in Neural Module Network","date":"2020-09-30","arxiv_id":"2009.14759","repositories_listed":0,"syntology":null},{"url":null,"slug":"strategy-and-benchmark-for-converting-deep-q","title":"Strategy and Benchmark for Converting Deep Q-Networks to Event-Driven Spiking Neural Networks","date":"2020-09-30","arxiv_id":"2009.14456","repositories_listed":0,"syntology":null},{"url":null,"slug":"teacher-critical-training-strategies-for","title":"Teacher-Critical Training Strategies for Image Captioning","date":"2020-09-30","arxiv_id":"2009.14405","repositories_listed":0,"syntology":null},{"url":null,"slug":"toolpath-design-for-additive-manufacturing","title":"Toolpath design for additive manufacturing using deep reinforcement learning","date":"2020-09-30","arxiv_id":"2009.14365","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-learning-in-deep-q-networks","title":"Cross Learning in Deep Q-Networks","date":"2020-09-29","arxiv_id":"2009.13780","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-design-space-adaptation-with-deep","title":"Trust-Region Method with Deep Reinforcement Learning in Analog Design Space Exploration","date":"2020-09-29","arxiv_id":"2009.13772","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-objective-reinforcement-learning-based","title":"Multi-objective Reinforcement Learning based approach for User-Centric Power Optimization in Smart Home Environments","date":"2020-09-29","arxiv_id":"2009.13854","repositories_listed":0,"syntology":null},{"url":null,"slug":"reannealing-of-decaying-exploration-based-on","title":"Reannealing of Decaying Exploration Based On Heuristic Measure in Deep Q-Network","date":"2020-09-29","arxiv_id":"2009.14297","repositories_listed":0,"syntology":null},{"url":null,"slug":"agent-environment-cycle-games","title":"Agent Environment Cycle Games","date":"2020-09-28","arxiv_id":"2009.13051","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-der-cyber","title":"Deep Reinforcement Learning for DER Cyber-Attack Mitigation","date":"2020-09-28","arxiv_id":"2009.13088","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-exploration-for-model-based","title":"Efficient Exploration for Model-based Reinforcement Learning with Continuous States and Actions","date":"2020-09-28","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"is-reinforcement-learning-more-difficult-than","title":"Is Reinforcement Learning More Difficult Than Bandits? A Near-optimal Algorithm Escaping the Curse of Horizon","date":"2020-09-28","arxiv_id":"2009.13503","repositories_listed":0,"syntology":null},{"url":null,"slug":"jointly-trained-state-action-embedding-for","title":"Jointly-Trained State-Action Embedding for Efficient Reinforcement Learning","date":"2020-09-28","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"mdp-playground-controlling-orthogonal","title":"MDP Playground: Controlling Orthogonal Dimensions of Hardness in Toy Environments","date":"2020-09-28","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"near-optimal-regret-bounds-for-model-free-rl","title":"Near-Optimal Regret Bounds for Model-Free RL in Non-Stationary Episodic MDPs","date":"2020-09-28","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"neuron-activation-analysis-for-multi-joint","title":"Neuron Activation Analysis for Multi-Joint Robot Reinforcement Learning","date":"2020-09-28","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-gradient-with-expected-quadratic","title":"Policy Gradient with Expected Quadratic Utility Maximization: A New Mean-Variance Approach in Reinforcement Learning","date":"2020-09-28","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"repaint-knowledge-transfer-in-deep-actor","title":"REPAINT: Knowledge Transfer in Deep Actor-Critic Reinforcement Learning","date":"2020-09-28","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-emergence-of-individuality-in-multi-agent-1","title":"The Emergence of Individuality in Multi-Agent Reinforcement Learning","date":"2020-09-28","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-heterogeneous-multi-agent","title":"Towards Heterogeneous Multi-Agent Reinforcement Learning with Graph Neural Networks","date":"2020-09-28","arxiv_id":"2009.13161","repositories_listed":0,"syntology":null},{"url":null,"slug":"transfer-among-agents-an-efficient-multiagent","title":"Transfer among Agents: An Efficient Multiagent Transfer Learning Framework","date":"2020-09-28","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"what-about-taking-policy-as-input-of-value","title":"What About Taking Policy as Input of Value Function: Policy-extended Value Function Approximator","date":"2020-09-28","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"machine-learning-in-event-triggered-control","title":"Machine Learning in Event-Triggered Control: Recent Advances and Open Issues","date":"2020-09-27","arxiv_id":"2009.12783","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-deep-reinforcement-learning-for-ride","title":"Scalable Deep Reinforcement Learning for Ride-Hailing","date":"2020-09-27","arxiv_id":"2009.14679","repositories_listed":0,"syntology":null},{"url":null,"slug":"scheduling-and-power-control-for-wireless","title":"Scheduling and Power Control for Wireless Multicast Systems via Deep Reinforcement Learning","date":"2020-09-27","arxiv_id":"2011.14799","repositories_listed":0,"syntology":null},{"url":null,"slug":"virtual-experience-to-real-world-application","title":"Virtual Experience to Real World Application: Sidewalk Obstacle Avoidance Using Reinforcement Learning for Visually Impaired","date":"2020-09-27","arxiv_id":"2009.12877","repositories_listed":0,"syntology":null},{"url":null,"slug":"complementary-meta-reinforcement-learning-for","title":"Complementary Meta-Reinforcement Learning for Fault-Adaptive Control","date":"2020-09-26","arxiv_id":"2009.12634","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-neural-induction-of-value-iteration","title":"Graph neural induction of value iteration","date":"2020-09-26","arxiv_id":"2009.12604","repositories_listed":0,"syntology":null},{"url":null,"slug":"inverse-rational-control-with-partially-1","title":"Inverse Rational Control with Partially Observable Continuous Nonlinear Dynamics","date":"2020-09-26","arxiv_id":"2009.12576","repositories_listed":0,"syntology":null},{"url":null,"slug":"lineage-evolution-reinforcement-learning","title":"Lineage Evolution Reinforcement Learning","date":"2020-09-26","arxiv_id":"2010.14616","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-n-ary-cross","title":"Reinforcement Learning-based N-ary Cross-Sentence Relation Extraction","date":"2020-09-26","arxiv_id":"2009.12683","repositories_listed":0,"syntology":null},{"url":null,"slug":"motion-planning-by-reinforcement-learning-for","title":"Motion Planning by Reinforcement Learning for an Unmanned Aerial Vehicle in Virtual Open Space with Static Obstacles","date":"2020-09-24","arxiv_id":"2009.11799","repositories_listed":0,"syntology":null},{"url":null,"slug":"sim-to-real-transfer-in-deep-reinforcement","title":"Sim-to-Real Transfer in Deep Reinforcement Learning for Robotics: a Survey","date":"2020-09-24","arxiv_id":"2009.13303","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multi-agent-deep-reinforcement-learning","title":"A Multi-Agent Deep Reinforcement Learning Approach for a Distributed Energy Marketplace in Smart Grids","date":"2020-09-23","arxiv_id":"2009.10905","repositories_listed":0,"syntology":null},{"url":null,"slug":"demand-responsive-dynamic-pricing-framework","title":"Demand Responsive Dynamic Pricing Framework for Prosumer Dominated Microgrids using Multiagent Reinforcement Learning","date":"2020-09-23","arxiv_id":"2009.10890","repositories_listed":0,"syntology":null},{"url":null,"slug":"probabilistic-machine-learning-for-healthcare","title":"Probabilistic Machine Learning for Healthcare","date":"2020-09-23","arxiv_id":"2009.11087","repositories_listed":0,"syntology":null},{"url":null,"slug":"releaser-a-reinforcement-learning-strategy","title":"ReLeaSER: A Reinforcement Learning Strategy for Optimizing Utilization Of Ephemeral Cloud Resources","date":"2020-09-23","arxiv_id":"2009.11208","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-reinforcement-learning-based","title":"Robust Reinforcement Learning-based Autonomous Driving Agent for Simulation and Real World","date":"2020-09-23","arxiv_id":"2009.11212","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-is-the-reward-for-handwriting","title":"What is the Reward for Handwriting? -- Handwriting Generation by Imitation Learning","date":"2020-09-23","arxiv_id":"2009.10962","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-on-line","title":"Deep Reinforcement Learning for On-line Dialogue State Tracking","date":"2020-09-22","arxiv_id":"2009.10321","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-structured-actor-critic","title":"Distributed Structured Actor-Critic Reinforcement Learning for Universal Dialogue Management","date":"2020-09-22","arxiv_id":"2009.10326","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-q-learning-provably-efficient-an-extended","title":"Is Q-Learning Provably Efficient? An Extended Analysis","date":"2020-09-22","arxiv_id":"2009.10396","repositories_listed":0,"syntology":null},{"url":null,"slug":"sumbt-larl-end-to-end-neural-task-oriented","title":"SUMBT+LaRL: Effective Multi-domain End-to-end Neural Task-oriented Dialog System","date":"2020-09-22","arxiv_id":"2009.10447","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-bandits-for-adapting-to-changing","title":"Contextual Bandits for adapting to changing User preferences over time","date":"2020-09-21","arxiv_id":"2009.10073","repositories_listed":0,"syntology":null},{"url":null,"slug":"dispatch-design-space-exploration-of-cyber","title":"DISPATCH: Design Space Exploration of Cyber-Physical Systems","date":"2020-09-21","arxiv_id":"2009.10214","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-horizon-value-estimation-for-model","title":"Dynamic Horizon Value Estimation for Model-based Reinforcement Learning","date":"2020-09-21","arxiv_id":"2009.09593","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-engagement-providing-evaluative-and","title":"Human Engagement Providing Evaluative and Informative Advice for Interactive Reinforcement Learning","date":"2020-09-21","arxiv_id":"2009.09575","repositories_listed":0,"syntology":null},{"url":null,"slug":"learn-to-exceed-stereo-inverse-reinforcement","title":"Learn to Exceed: Stereo Inverse Reinforcement Learning with Concurrent Policy Optimization","date":"2020-09-21","arxiv_id":"2009.09577","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-a-contact-adaptive-controller-for","title":"Learning a Contact-Adaptive Controller for Robust, Efficient Legged Locomotion","date":"2020-09-21","arxiv_id":"2009.10019","repositories_listed":0,"syntology":null},{"url":null,"slug":"mobile-cellular-connected-uavs-reinforcement","title":"Mobile Cellular-Connected UAVs: Reinforcement Learning for Sky Limits","date":"2020-09-21","arxiv_id":"2009.09815","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-approaches-in-social","title":"Reinforcement Learning Approaches in Social Robotics","date":"2020-09-21","arxiv_id":"2009.09689","repositories_listed":0,"syntology":null},{"url":null,"slug":"lyapunov-based-reinforcement-learning-for","title":"Lyapunov-Based Reinforcement Learning for Decentralized Multi-Agent Control","date":"2020-09-20","arxiv_id":"2009.09361","repositories_listed":0,"syntology":null},{"url":null,"slug":"multiplayer-support-for-the-arcade-learning","title":"Multiplayer Support for the Arcade Learning Environment","date":"2020-09-20","arxiv_id":"2009.09341","repositories_listed":0,"syntology":null},{"url":null,"slug":"regret-bounds-and-reinforcement-learning","title":"Regret Bounds and Reinforcement Learning Exploration of EXP-based Algorithms","date":"2020-09-20","arxiv_id":"2009.09538","repositories_listed":0,"syntology":null},{"url":null,"slug":"construction-of-polar-codes-with","title":"Construction of Polar Codes with Reinforcement Learning","date":"2020-09-19","arxiv_id":"2009.09277","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-contraction-approach-to-model-based","title":"A Contraction Approach to Model-based Reinforcement Learning","date":"2020-09-18","arxiv_id":"2009.08586","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-closed-loop","title":"Deep Reinforcement Learning for Closed-Loop Blood Glucose Control","date":"2020-09-18","arxiv_id":"2009.09051","repositories_listed":0,"syntology":null},{"url":null,"slug":"private-reinforcement-learning-with-pac-and","title":"Private Reinforcement Learning with PAC and Regret Guarantees","date":"2020-09-18","arxiv_id":"2009.09052","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-weakly-supervised","title":"Reinforcement Learning for Weakly Supervised Temporal Grounding of Natural Language in Untrimmed Videos","date":"2020-09-18","arxiv_id":"2009.08614","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalight-improving-environment","title":"GeneraLight: Improving Environment Generalization of Traffic Signal Control via Meta Reinforcement Learning","date":"2020-09-17","arxiv_id":"2009.08052","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-assisted-deep-reinforcement","title":"Knowledge-Assisted Deep Reinforcement Learning in 5G Scheduler Design: From Theoretical Framework to Implementation","date":"2020-09-17","arxiv_id":"2009.08346","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-relationship-between-dynamic-programming","title":"Reward Maximisation through Discrete Active Inference","date":"2020-09-17","arxiv_id":"2009.08111","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-behavior-level-explanation-for-deep","title":"Reconstructing Actions To Explain Deep Reinforcement Learning","date":"2020-09-17","arxiv_id":"2009.08507","repositories_listed":0,"syntology":null},{"url":null,"slug":"drl-fas-a-novel-framework-based-on-deep","title":"DRL-FAS: A Novel Framework Based on Deep Reinforcement Learning for Face Anti-Spoofing","date":"2020-09-16","arxiv_id":"2009.07529","repositories_listed":0,"syntology":null},{"url":null,"slug":"theory-of-mind-with-guilt-aversion","title":"Theory of Mind with Guilt Aversion Facilitates Cooperative Reinforcement Learning","date":"2020-09-16","arxiv_id":"2009.07445","repositories_listed":0,"syntology":null},{"url":null,"slug":"time-your-hedge-with-deep-reinforcement","title":"Time your hedge with Deep Reinforcement Learning","date":"2020-09-16","arxiv_id":"2009.14136","repositories_listed":0,"syntology":null},{"url":null,"slug":"transfer-learning-in-deep-reinforcement","title":"Transfer Learning in Deep Reinforcement Learning: A Survey","date":"2020-09-16","arxiv_id":"2009.07888","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-learning-of-features-for-control","title":"Autonomous Learning of Features for Control: Experiments with Embodied and Situated Agents","date":"2020-09-15","arxiv_id":"2009.07132","repositories_listed":0,"syntology":null},{"url":null,"slug":"decoding-polar-codes-with-reinforcement","title":"Decoding Polar Codes with Reinforcement Learning","date":"2020-09-15","arxiv_id":"2009.06796","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-strategic","title":"Reinforcement Learning for Strategic Recommendations","date":"2020-09-15","arxiv_id":"2009.07346","repositories_listed":0,"syntology":null},{"url":null,"slug":"soft-policy-optimization-using-dual-track","title":"Soft policy optimization using dual-track advantage estimator","date":"2020-09-15","arxiv_id":"2009.06858","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-transformers-a-survey","title":"Efficient Transformers: A Survey","date":"2020-09-14","arxiv_id":"2009.06732","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-in-cournot","title":"Multi-Agent Reinforcement Learning in Cournot Games","date":"2020-09-14","arxiv_id":"2009.06224","repositories_listed":0,"syntology":null},{"url":null,"slug":"predictive-synthesis-of-quantum-materials-by","title":"Predictive Synthesis of Quantum Materials by Probabilistic Reinforcement Learning","date":"2020-09-14","arxiv_id":"2009.06739","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-dynamic-resource","title":"Reinforcement Learning for Dynamic Resource Optimization in 5G Radio Access Network Slicing","date":"2020-09-14","arxiv_id":"2009.06579","repositories_listed":0,"syntology":null},{"url":null,"slug":"variance-reduced-off-policy-memory-efficient","title":"Variance-Reduced Off-Policy Memory-Efficient Policy Search","date":"2020-09-14","arxiv_id":"2009.06548","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-competitive-self-play-policy","title":"Efficient Competitive Self-Play Policy Optimization","date":"2020-09-13","arxiv_id":"2009.06086","repositories_listed":0,"syntology":null},{"url":null,"slug":"extended-radial-basis-function-controller-for","title":"Extended Radial Basis Function Controller for Reinforcement Learning","date":"2020-09-12","arxiv_id":"2009.05866","repositories_listed":0,"syntology":null}],"record_sha256":"15a9225d4fec6a17c22e66b973af14e55bf3c5799a3c20071ce389e02a7e0605","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}