{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/90","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":90,"pages_in_order":132,"rows_per_page":100,"rows":[8901,9000],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/89","next":"/task/reinforcement-learning/papers/91","papers":[{"url":null,"slug":"reinforcement-learning-with-efficient-active","title":"Reinforcement Learning with Efficient Active Feature Acquisition","date":"2020-11-02","arxiv_id":"2011.00825","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-efficient-reinforcement-learning-using","title":"Sample-efficient reinforcement learning using deep Gaussian processes","date":"2020-11-02","arxiv_id":"2011.01226","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-imbalanced","title":"Reinforcement Learning with Imbalanced Dataset for Data-to-Text Medical Report Generation","date":"2020-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"abstract-value-iteration-for-hierarchical","title":"Abstract Value Iteration for Hierarchical Reinforcement Learning","date":"2020-10-29","arxiv_id":"2010.15638","repositories_listed":0,"syntology":null},{"url":null,"slug":"causal-variables-from-reinforcement-learning","title":"Reinforcement Learning of Causal Variables Using Mediation Analysis","date":"2020-10-29","arxiv_id":"2010.15745","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-do-offline-measures-for-exploration-in","title":"How do Offline Measures for Exploration in Reinforcement Learning behave?","date":"2020-10-29","arxiv_id":"2010.15533","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-versus-machine-attention-in-deep","title":"Machine versus Human Attention in Deep Reinforcement Learning Tasks","date":"2020-10-29","arxiv_id":"2010.15942","repositories_listed":0,"syntology":null},{"url":null,"slug":"deepfoldit-a-deep-reinforcement-learning","title":"DeepFoldit -- A Deep Reinforcement Learning Neural Network Folding Proteins","date":"2020-10-28","arxiv_id":"2011.03442","repositories_listed":0,"syntology":null},{"url":null,"slug":"designing-interpretable-approximations-to","title":"Designing Interpretable Approximations to Deep Reinforcement Learning","date":"2020-10-28","arxiv_id":"2010.14785","repositories_listed":0,"syntology":null},{"url":null,"slug":"batch-reinforcement-learning-with-a","title":"Batch Reinforcement Learning with a Nonparametric Off-Policy Policy Gradient","date":"2020-10-27","arxiv_id":"2010.14771","repositories_listed":0,"syntology":null},{"url":null,"slug":"behavior-priors-for-efficient-reinforcement","title":"Behavior Priors for Efficient Reinforcement Learning","date":"2020-10-27","arxiv_id":"2010.14274","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-reinforcement-learning-for-continuous","title":"Can Reinforcement Learning for Continuous Control Generalize Across Physics Engines?","date":"2020-10-27","arxiv_id":"2010.14444","repositories_listed":0,"syntology":null},{"url":null,"slug":"behavioral-decision-making-for-urban","title":"Behavioral decision-making for urban autonomous driving in the presence of pedestrians using Deep Recurrent Q-Network","date":"2020-10-26","arxiv_id":"2010.13407","repositories_listed":0,"syntology":null},{"url":null,"slug":"lyapunov-based-reinforcement-learning-state","title":"Lyapunov-Based Reinforcement Learning State Estimator","date":"2020-10-26","arxiv_id":"2010.13529","repositories_listed":0,"syntology":null},{"url":null,"slug":"opal-offline-primitive-discovery-for-1","title":"OPAL: Offline Primitive Discovery for Accelerating Offline Reinforcement Learning","date":"2020-10-26","arxiv_id":"2010.13611","repositories_listed":0,"syntology":null},{"url":null,"slug":"pairwise-heuristic-sequence-alignment","title":"Pairwise heuristic sequence alignment algorithm based on deep reinforcement learning","date":"2020-10-26","arxiv_id":"2010.13478","repositories_listed":0,"syntology":null},{"url":null,"slug":"visualhints-a-visual-lingual-environment-for","title":"VisualHints: A Visual-Lingual Environment for Multimodal Reinforcement Learning","date":"2020-10-26","arxiv_id":"2010.13839","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-the-exploration-of-deep","title":"Improving the Exploration of Deep Reinforcement Learning in Continuous Domains using Planning for Policy Search","date":"2020-10-24","arxiv_id":"2010.12974","repositories_listed":0,"syntology":null},{"url":null,"slug":"planning-with-exploration-addressing-dynamics","title":"Planning with Exploration: Addressing Dynamics Bottleneck in Model-based Reinforcement Learning","date":"2020-10-24","arxiv_id":"2010.12914","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-worst-case-regret-bounds-for","title":"Improved Worst-Case Regret Bounds for Randomized Least-Squares Value Iteration","date":"2020-10-23","arxiv_id":"2010.12163","repositories_listed":0,"syntology":null},{"url":null,"slug":"option-hedging-with-risk-averse-reinforcement","title":"Option Hedging with Risk Averse Reinforcement Learning","date":"2020-10-23","arxiv_id":"2010.12245","repositories_listed":0,"syntology":null},{"url":null,"slug":"stochastic-inverse-reinforcement-learning-1","title":"Stochastic Inverse Reinforcement Learning","date":"2020-10-23","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"error-bounds-of-imitating-policies-and","title":"Error Bounds of Imitating Policies and Environments","date":"2020-10-22","arxiv_id":"2010.11876","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimising-stochastic-routing-for-taxi-fleets","title":"Optimising Stochastic Routing for Taxi Fleets with Model Enhanced Reinforcement Learning","date":"2020-10-22","arxiv_id":"2010.11738","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-efficient-reinforcement-learning-with-1","title":"Sample Efficient Reinforcement Learning with REINFORCE","date":"2020-10-22","arxiv_id":"2010.11364","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-are-the-statistical-limits-of-offline-rl","title":"What are the Statistical Limits of Offline RL with Linear Function Approximation?","date":"2020-10-22","arxiv_id":"2010.11895","repositories_listed":0,"syntology":null},{"url":null,"slug":"safety-verification-of-model-based-1","title":"Safety Verification of Model Based Reinforcement Learning Controllers","date":"2020-10-21","arxiv_id":"2010.10740","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-in-lane-merge","title":"Deep Reinforcement Learning in Lane Merge Coordination for Connected Vehicles","date":"2020-10-20","arxiv_id":"2010.10567","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-inference-with-multi-head-automata","title":"Language Inference with Multi-head Automata through Reinforcement Learning","date":"2020-10-20","arxiv_id":"2010.10141","repositories_listed":0,"syntology":null},{"url":null,"slug":"negotiating-team-formation-using-deep-1","title":"Negotiating Team Formation Using Deep Reinforcement Learning","date":"2020-10-20","arxiv_id":"2010.10380","repositories_listed":0,"syntology":null},{"url":null,"slug":"quality-of-service-based-radar-resource","title":"Quality of service based radar resource management using deep reinforcement learning","date":"2020-10-20","arxiv_id":"2010.10210","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-constrained-reinforcement-learning-for-1","title":"Robust Constrained Reinforcement Learning for Continuous Control with Model Misspecification","date":"2020-10-20","arxiv_id":"2010.10644","repositories_listed":0,"syntology":null},{"url":null,"slug":"runtime-safety-assurance-using-reinforcement","title":"Runtime Safety Assurance Using Reinforcement Learning","date":"2020-10-20","arxiv_id":"2010.10618","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-approach-to-health","title":"A Reinforcement Learning Approach to Health Aware Control Strategy","date":"2020-10-19","arxiv_id":"2010.09269","repositories_listed":0,"syntology":null},{"url":null,"slug":"chance-constrained-control-with-lexicographic","title":"Chance-Constrained Control with Lexicographic Deep Reinforcement Learning","date":"2020-10-19","arxiv_id":"2010.09468","repositories_listed":0,"syntology":null},{"url":null,"slug":"imitation-with-neural-density-models-1","title":"Imitation with Neural Density Models","date":"2020-10-19","arxiv_id":"2010.09808","repositories_listed":0,"syntology":null},{"url":null,"slug":"average-reward-model-free-reinforcement","title":"Average-reward model-free reinforcement learning: a systematic review and literature mapping","date":"2020-10-18","arxiv_id":"2010.08920","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-inverse-reinforcement-learning","title":"Model-Based Inverse Reinforcement Learning from Visual Demonstrations","date":"2020-10-18","arxiv_id":"2010.09034","repositories_listed":0,"syntology":null},{"url":null,"slug":"assessment-of-reward-functions-in","title":"Assessment of Reward Functions in Reinforcement Learning for Multi-Modal Urban Traffic Control under Real-World limitations","date":"2020-10-17","arxiv_id":"2010.08819","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-elimination-ordering-for-tree","title":"Learning Elimination Ordering for Tree Decomposition Problem","date":"2020-10-17","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-lower-bounds-for-graph-exploration","title":"Learning Lower Bounds for Graph Exploration With Reinforcement Learning","date":"2020-10-17","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"variational-dynamic-for-self-supervised","title":"Variational Dynamic for Self-Supervised Exploration in Deep Reinforcement Learning","date":"2020-10-17","arxiv_id":"2010.08755","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-control-of-a-particle-accelerator","title":"Autonomous Control of a Particle Accelerator using Deep Reinforcement Learning","date":"2020-10-16","arxiv_id":"2010.08141","repositories_listed":0,"syntology":null},{"url":null,"slug":"doom-a-novel-adversarial-drl-based-op-code","title":"DOOM: A Novel Adversarial-DRL-Based Op-Code Level Metamorphic Malware Obfuscator for the Enhancement of IDS","date":"2020-10-16","arxiv_id":"2010.08608","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-robotic-object-search-via-hiem","title":"Efficient Robotic Object Search via HIEM: Hierarchical Policy Learning with Intrinsic-Extrinsic Modeling","date":"2020-10-16","arxiv_id":"2010.08596","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-efficient-and","title":"Reinforcement Learning for Efficient and Tuning-Free Link Adaptation","date":"2020-10-16","arxiv_id":"2010.08651","repositories_listed":0,"syntology":null},{"url":null,"slug":"uncertainty-aware-contact-safe-model-based","title":"Uncertainty-aware Contact-safe Model-based Reinforcement Learning","date":"2020-10-16","arxiv_id":"2010.08169","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-nesterov-s-accelerated-quasi-newton-method","title":"A Nesterov's Accelerated quasi-Newton method for Global Routing using Deep Reinforcement Learning","date":"2020-10-15","arxiv_id":"2010.09465","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-empowerment-based-solution-to-robotic","title":"An Empowerment-based Solution to Robotic Manipulation Tasks with Sparse Rewards","date":"2020-10-15","arxiv_id":"2010.07986","repositories_listed":0,"syntology":null},{"url":null,"slug":"blending-search-and-discovery-tag-based-query","title":"Blending Search and Discovery: Tag-Based Query Refinement with Contextual Reinforcement Learning","date":"2020-10-15","arxiv_id":"2010.09495","repositories_listed":0,"syntology":null},{"url":null,"slug":"cooperative-competitive-reinforcement","title":"Cooperative-Competitive Reinforcement Learning with History-Dependent Rewards","date":"2020-10-15","arxiv_id":"2010.08030","repositories_listed":0,"syntology":null},{"url":null,"slug":"explanation-augmented-feedback-in-human-in-1","title":"Explanation Augmented Feedback in Human-in-the-Loop Reinforcement Learning","date":"2020-10-15","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"local-differentially-private-regret","title":"Local Differential Privacy for Regret Minimization in Reinforcement Learning","date":"2020-10-15","arxiv_id":"2010.07778","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-dispatch-in-emergency-service-system","title":"Optimal Dispatch in Emergency Service System via Reinforcement Learning","date":"2020-10-15","arxiv_id":"2010.07513","repositories_listed":0,"syntology":null},{"url":null,"slug":"average-cost-optimal-control-of-stochastic","title":"Average Cost Optimal Control of Stochastic Systems Using Reinforcement Learning","date":"2020-10-13","arxiv_id":"2010.06236","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-and","title":"Deep Reinforcement Learning and Transportation Research: A Comprehensive Review","date":"2020-10-13","arxiv_id":"2010.06187","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-reinforcement-learning-for-type","title":"Model-Based Reinforcement Learning for Type 1Diabetes Blood Glucose Control","date":"2020-10-13","arxiv_id":"2010.06266","repositories_listed":0,"syntology":null},{"url":"/paper/discrete-latent-space-world-models-for","slug":"discrete-latent-space-world-models-for","title":"Smaller World Models for Reinforcement Learning","date":"2020-10-12","arxiv_id":"2010.05767","repositories_listed":0,"syntology":null},{"url":null,"slug":"nearly-minimax-optimal-reward-free","title":"Nearly Minimax Optimal Reward-free Reinforcement Learning","date":"2020-10-12","arxiv_id":"2010.05901","repositories_listed":0,"syntology":null},{"url":null,"slug":"remote-electrical-tilt-optimization-via-safe","title":"Remote Electrical Tilt Optimization via Safe Reinforcement Learning","date":"2020-10-12","arxiv_id":"2010.05842","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-to-stop-epidemics-controlling-graph","title":"Controlling Graph Dynamics with Reinforcement Learning and Graph Neural Networks","date":"2020-10-11","arxiv_id":"2010.05313","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-with-natural-1","title":"Safe Reinforcement Learning with Natural Language Constraints","date":"2020-10-11","arxiv_id":"2010.05150","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-on-computational","title":"Reinforcement Learning on Computational Resource Allocation of Cloud-based Wireless Networks","date":"2020-10-10","arxiv_id":"2010.05024","repositories_listed":0,"syntology":null},{"url":null,"slug":"characterizing-policy-divergence-for","title":"Characterizing Policy Divergence for Personalized Meta-Reinforcement Learning","date":"2020-10-09","arxiv_id":"2010.04816","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-asset","title":"Deep Reinforcement Learning for Asset Allocation in US Equities","date":"2020-10-09","arxiv_id":"2010.04404","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-rl-with-information-constrained-policies","title":"Deep RL With Information Constrained Policies: Generalization in Continuous Control","date":"2020-10-09","arxiv_id":"2010.04646","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-state-action-embedding-for-efficient-1","title":"Jointly-Learned State-Action Embedding for Efficient Reinforcement Learning","date":"2020-10-09","arxiv_id":"2010.04444","repositories_listed":0,"syntology":null},{"url":null,"slug":"parameterized-reinforcement-learning-for","title":"Parameterized Reinforcement Learning for Optical System Optimization","date":"2020-10-09","arxiv_id":"2010.05769","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-intrinsic-symbolic-rewards-in-1","title":"Learning Intrinsic Symbolic Rewards in Reinforcement Learning","date":"2020-10-08","arxiv_id":"2010.03694","repositories_listed":0,"syntology":null},{"url":null,"slug":"nonstationary-reinforcement-learning-with","title":"Nonstationary Reinforcement Learning with Linear Function Approximation","date":"2020-10-08","arxiv_id":"2010.04244","repositories_listed":0,"syntology":null},{"url":null,"slug":"provable-fictitious-play-for-general-mean-1","title":"Provable Fictitious Play for General Mean-Field Games","date":"2020-10-08","arxiv_id":"2010.04211","repositories_listed":0,"syntology":null},{"url":null,"slug":"episodic-reinforcement-learning-in-finite","title":"Episodic Reinforcement Learning in Finite MDPs: Minimax Lower Bounds Revisited","date":"2020-10-07","arxiv_id":"2010.03531","repositories_listed":0,"syntology":null},{"url":null,"slug":"instance-dependent-complexity-of-contextual","title":"Instance-Dependent Complexity of Contextual Bandits and Reinforcement Learning: A Disagreement-Based Perspective","date":"2020-10-07","arxiv_id":"2010.03104","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-safety-assurance-for-deep","title":"Online Safety Assurance for Deep Reinforcement Learning","date":"2020-10-07","arxiv_id":"2010.03625","repositories_listed":0,"syntology":null},{"url":null,"slug":"regularized-inverse-reinforcement-learning-1","title":"Regularized Inverse Reinforcement Learning","date":"2020-10-07","arxiv_id":"2010.03691","repositories_listed":0,"syntology":null},{"url":null,"slug":"heterogeneous-multi-agent-reinforcement","title":"Heterogeneous Multi-Agent Reinforcement Learning for Unknown Environment Mapping","date":"2020-10-06","arxiv_id":"2010.02663","repositories_listed":0,"syntology":null},{"url":null,"slug":"safety-aware-reinforcement-learning-sarl-1","title":"Safety Aware Reinforcement Learning (SARL)","date":"2020-10-06","arxiv_id":"2010.02846","repositories_listed":0,"syntology":null},{"url":null,"slug":"uneven-universal-value-exploration-for-multi-1","title":"UneVEn: Universal Value Exploration for Multi-Agent Reinforcement Learning","date":"2020-10-06","arxiv_id":"2010.02974","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-collaborative","title":"Deep Reinforcement Learning for Collaborative Edge Computing in Vehicular Networks","date":"2020-10-05","arxiv_id":"2010.01722","repositories_listed":0,"syntology":null},{"url":null,"slug":"goal-directed-generation-of-discrete","title":"Goal-directed Generation of Discrete Structures with Conditional Generative Models","date":"2020-10-05","arxiv_id":"2010.02311","repositories_listed":0,"syntology":null},{"url":null,"slug":"sentiment-analysis-for-reinforcement-learning","title":"Sentiment Analysis for Reinforcement Learning","date":"2020-10-05","arxiv_id":"2010.02316","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-sharp-analysis-of-model-based-reinforcement-1","title":"A Sharp Analysis of Model-based Reinforcement Learning with Self-Play","date":"2020-10-04","arxiv_id":"2010.01604","repositories_listed":0,"syntology":null},{"url":null,"slug":"attractor-selection-in-nonlinear-energy","title":"Attractor Selection in Nonlinear Energy Harvesting Using Deep Reinforcement Learning","date":"2020-10-03","arxiv_id":"2010.01255","repositories_listed":0,"syntology":null},{"url":null,"slug":"disentangling-causal-effects-for-hierarchical","title":"Disentangling causal effects for hierarchical reinforcement learning","date":"2020-10-03","arxiv_id":"2010.01351","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-gradient-with-expected-quadratic-1","title":"Mean-Variance Efficient Reinforcement Learning with Applications to Dynamic Financial Investment","date":"2020-10-03","arxiv_id":"2010.01404","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-of-simple-indirect","title":"Reinforcement Learning of Sequential Price Mechanisms","date":"2020-10-02","arxiv_id":"2010.01180","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-mixed","title":"Deep Reinforcement Learning with Mixed Convolutional Network","date":"2020-10-01","arxiv_id":"2010.00717","repositories_listed":0,"syntology":null},{"url":null,"slug":"minimax-optimal-reinforcement-learning-for","title":"Nearly Minimax Optimal Reinforcement Learning for Discounted MDPs","date":"2020-10-01","arxiv_id":"2010.00587","repositories_listed":0,"syntology":null},{"url":"/paper/multi-agent-social-reinforcement-learning","slug":"multi-agent-social-reinforcement-learning","title":"Emergent Social Learning via Multi-agent Reinforcement Learning","date":"2020-10-01","arxiv_id":"2010.00581","repositories_listed":0,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multi-agent-social-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2010.00581","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.00581"}},"official":null}},{"url":null,"slug":"multi-reward-based-reinforcement-learning-for","title":"Multi-Reward based Reinforcement Learning for Neural Machine Translation","date":"2020-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"recognition-method-of-important-words-in","title":"Recognition Method of Important Words in Korean Text based on Reinforcement Learning","date":"2020-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"value-based-bayesian-meta-reinforcement","title":"Bayesian Meta-reinforcement Learning for Traffic Signal Control","date":"2020-10-01","arxiv_id":"2010.00163","repositories_listed":0,"syntology":null},{"url":null,"slug":"aamdrl-augmented-asset-management-with-deep","title":"AAMDRL: Augmented Asset Management with Deep Reinforcement Learning","date":"2020-09-30","arxiv_id":"2010.08497","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-optimization-and-reinforcement","title":"Accelerating Optimization and Reinforcement Learning with Quasi-Stochastic Approximation","date":"2020-09-30","arxiv_id":"2009.14431","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-the-gap-between-markowitz-planning","title":"Bridging the gap between Markowitz planning and deep reinforcement learning","date":"2020-09-30","arxiv_id":"2010.09108","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-based-heuristic-search-for-module","title":"Graph-based Heuristic Search for Module Selection Procedure in Neural Module Network","date":"2020-09-30","arxiv_id":"2009.14759","repositories_listed":0,"syntology":null},{"url":null,"slug":"strategy-and-benchmark-for-converting-deep-q","title":"Strategy and Benchmark for Converting Deep Q-Networks to Event-Driven Spiking Neural Networks","date":"2020-09-30","arxiv_id":"2009.14456","repositories_listed":0,"syntology":null},{"url":null,"slug":"toolpath-design-for-additive-manufacturing","title":"Toolpath design for additive manufacturing using deep reinforcement learning","date":"2020-09-30","arxiv_id":"2009.14365","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-der-cyber","title":"Deep Reinforcement Learning for DER Cyber-Attack Mitigation","date":"2020-09-28","arxiv_id":"2009.13088","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-reinforcement-learning-more-difficult-than","title":"Is Reinforcement Learning More Difficult Than Bandits? A Near-optimal Algorithm Escaping the Curse of Horizon","date":"2020-09-28","arxiv_id":"2009.13503","repositories_listed":0,"syntology":null}],"record_sha256":"2e7828f349ef4676b4dd3fbf0988005618adb513028bfcd5c9598fa157bf4a1f","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}