{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/86","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":86,"pages_in_order":152,"rows_per_page":100,"rows":[8501,8600],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/85","next":"/task/reinforcement-learning-1/papers/87","papers":[{"url":null,"slug":"learning-open-domain-multi-hop-search-using","title":"Learning Open Domain Multi-hop Search Using Reinforcement Learning","date":"2022-05-30","arxiv_id":"2205.15281","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantum-multi-armed-bandits-and-stochastic","title":"Quantum Multi-Armed Bandits and Stochastic Linear Bandits Enjoy Logarithmic Regrets","date":"2022-05-30","arxiv_id":"2205.14988","repositories_listed":0,"syntology":null},{"url":null,"slug":"residual-q-networks-for-value-function","title":"Residual Q-Networks for Value Function Factorizing in Multi-Agent Reinforcement Learning","date":"2022-05-30","arxiv_id":"2205.15245","repositories_listed":0,"syntology":null},{"url":null,"slug":"seren-knowing-when-to-explore-and-when-to","title":"SEREN: Knowing When to Explore and When to Exploit","date":"2022-05-30","arxiv_id":"2205.15064","repositories_listed":0,"syntology":null},{"url":null,"slug":"stock-trading-optimization-through-model","title":"Stock Trading Optimization through Model-based Reinforcement Learning with Resistance Support Relative Strength","date":"2022-05-30","arxiv_id":"2205.15056","repositories_listed":0,"syntology":null},{"url":null,"slug":"representation-gap-in-deep-reinforcement","title":"Frustratingly Easy Regularization on Representation Can Boost Deep Reinforcement Learning","date":"2022-05-29","arxiv_id":"2205.14557","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-source-transfer-learning-for-deep-model","title":"Multi-Source Transfer Learning for Deep Model-Based Reinforcement Learning","date":"2022-05-28","arxiv_id":"2205.14410","repositories_listed":0,"syntology":null},{"url":null,"slug":"survival-analysis-on-structured-data-using","title":"Survival Analysis on Structured Data using Deep Reinforcement Learning","date":"2022-05-28","arxiv_id":"2205.14331","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-distributed-1","title":"Deep Reinforcement Learning for Distributed and Uncoordinated Cognitive Radios Resource Allocation","date":"2022-05-27","arxiv_id":"2205.13944","repositories_listed":0,"syntology":null},{"url":null,"slug":"double-deep-q-networks-for-sensor-management","title":"Double Deep Q Networks for Sensor Management in Space Situational Awareness","date":"2022-05-27","arxiv_id":"2205.14041","repositories_listed":0,"syntology":null},{"url":null,"slug":"feudal-multi-agent-reinforcement-learning","title":"Feudal Multi-Agent Reinforcement Learning with Adaptive Network Partition for Traffic Signal Control","date":"2022-05-27","arxiv_id":"2205.13836","repositories_listed":0,"syntology":null},{"url":null,"slug":"galois-boosting-deep-reinforcement-learning","title":"GALOIS: Boosting Deep Reinforcement Learning via Generalizable Logic Synthesis","date":"2022-05-27","arxiv_id":"2205.13728","repositories_listed":0,"syntology":null},{"url":null,"slug":"kl-entropy-regularized-rl-with-a-generative","title":"KL-Entropy-Regularized RL with a Generative Model is Minimax Optimal","date":"2022-05-27","arxiv_id":"2205.14211","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-markovian-policies-occupancy-measures","title":"Non-Markovian policies occupancy measures","date":"2022-05-27","arxiv_id":"2205.13950","repositories_listed":0,"syntology":null},{"url":null,"slug":"off-beat-multi-agent-reinforcement-learning","title":"Off-Beat Multi-Agent Reinforcement Learning","date":"2022-05-27","arxiv_id":"2205.13718","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-sample-efficient-rl-with-side","title":"Provably Sample-Efficient RL with Side Information about Latent Dynamics","date":"2022-05-27","arxiv_id":"2205.14237","repositories_listed":0,"syntology":null},{"url":null,"slug":"tutorial-on-course-of-action-coa-attack","title":"Tutorial on Course-of-Action (COA) Attack Search Methods in Computer Networks","date":"2022-05-27","arxiv_id":"2205.13763","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-fair-federated-learning-framework-with","title":"A Fair Federated Learning Framework With Reinforcement Learning","date":"2022-05-26","arxiv_id":"2205.13415","repositories_listed":0,"syntology":null},{"url":null,"slug":"approximate-q-learning-and-sarsa-0-under-the","title":"Does DQN Learn?","date":"2022-05-26","arxiv_id":"2205.13617","repositories_listed":0,"syntology":null},{"url":null,"slug":"constrained-reinforcement-learning-for-short","title":"Constrained Reinforcement Learning for Short Video Recommendation","date":"2022-05-26","arxiv_id":"2205.13248","repositories_listed":0,"syntology":null},{"url":null,"slug":"embed-to-control-partially-observed-systems","title":"Embed to Control Partially Observed Systems: Representation Learning with Provable Sample Efficiency","date":"2022-05-26","arxiv_id":"2205.13476","repositories_listed":0,"syntology":null},{"url":null,"slug":"pessimism-in-the-face-of-confounders-provably","title":"Pessimism in the Face of Confounders: Provably Efficient Offline Reinforcement Learning in Partially Observable Markov Decision Processes","date":"2022-05-26","arxiv_id":"2205.13589","repositories_listed":0,"syntology":null},{"url":null,"slug":"physics-guided-hierarchical-reward-mechanism","title":"Physics-Guided Hierarchical Reward Mechanism for Learning-Based Robotic Grasping","date":"2022-05-26","arxiv_id":"2205.13561","repositories_listed":0,"syntology":null},{"url":null,"slug":"race-a-reinforcement-learning-framework-for","title":"RACE: A Reinforcement Learning Framework for Improved Adaptive Control of NoC Channel Buffers","date":"2022-05-26","arxiv_id":"2205.13130","repositories_listed":0,"syntology":null},{"url":null,"slug":"temporl-temporal-priors-for-exploration-in-1","title":"SFP: State-free Priors for Exploration in Off-Policy Reinforcement Learning","date":"2022-05-26","arxiv_id":"2205.13528","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-experimental-comparison-between-temporal","title":"An Experimental Comparison Between Temporal Difference and Residual Gradient with Neural Network Approximation","date":"2022-05-25","arxiv_id":"2205.12770","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-inference-and-transfer-of-compositional","title":"Fast Inference and Transfer of Compositional Task Structures for Few-shot Task Generalization","date":"2022-05-25","arxiv_id":"2205.12648","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-mean-field-games-a-survey","title":"Learning in Mean Field Games: A Survey","date":"2022-05-25","arxiv_id":"2205.12944","repositories_listed":0,"syntology":null},{"url":null,"slug":"maviper-learning-decision-tree-policies-for","title":"MAVIPER: Learning Decision Tree Policies for Interpretable Multi-Agent Reinforcement Learning","date":"2022-05-25","arxiv_id":"2205.12449","repositories_listed":0,"syntology":null},{"url":null,"slug":"near-optimal-goal-oriented-reinforcement","title":"Near-Optimal Goal-Oriented Reinforcement Learning in Non-Stationary Environments","date":"2022-05-25","arxiv_id":"2205.13044","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-reinforcement-learning-on-graphs-for","title":"Robust Reinforcement Learning on Graphs for Logistics optimization","date":"2022-05-25","arxiv_id":"2205.12888","repositories_listed":0,"syntology":null},{"url":null,"slug":"stochastic-second-order-methods-provably-beat","title":"Stochastic Second-Order Methods Improve Best-Known Sample Complexity of SGD for Gradient-Dominated Function","date":"2022-05-25","arxiv_id":"2205.12856","repositories_listed":0,"syntology":null},{"url":null,"slug":"transportation-inequalities-lyapunov","title":"Transportation-Inequalities, Lyapunov Stability and Sampling for Dynamical Systems on Continuous State Space","date":"2022-05-25","arxiv_id":"2205.12448","repositories_listed":0,"syntology":null},{"url":null,"slug":"trust-based-consensus-in-multi-agent","title":"Trust-based Consensus in Multi-Agent Reinforcement Learning Systems","date":"2022-05-25","arxiv_id":"2205.12880","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributional-hamilton-jacobi-bellman","title":"Distributional Hamilton-Jacobi-Bellman Equations for Continuous-Time Reinforcement Learning","date":"2022-05-24","arxiv_id":"2205.12184","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-convolutional-reinforcement-learning-1","title":"Graph Convolutional Reinforcement Learning for Collaborative Queuing Agents","date":"2022-05-24","arxiv_id":"2205.12009","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-drive-using-sparse-imitation","title":"Learning to Drive Using Sparse Imitation Reinforcement Learning","date":"2022-05-24","arxiv_id":"2205.12128","repositories_listed":0,"syntology":null},{"url":null,"slug":"penalized-proximal-policy-optimization-for","title":"Penalized Proximal Policy Optimization for Safe Reinforcement Learning","date":"2022-05-24","arxiv_id":"2205.11814","repositories_listed":0,"syntology":null},{"url":null,"slug":"computationally-efficient-horizon-free","title":"Computationally Efficient Horizon-Free Reinforcement Learning for Linear Mixture MDPs","date":"2022-05-23","arxiv_id":"2205.11507","repositories_listed":0,"syntology":null},{"url":null,"slug":"cooperative-reinforcement-learning-on-traffic","title":"Cooperative Reinforcement Learning on Traffic Signal Control","date":"2022-05-23","arxiv_id":"2205.11291","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-reinforcement-learning-from","title":"Efficient Reinforcement Learning from Demonstration Using Local Ensemble and Reparameterization with Split and Merge of Expert Policies","date":"2022-05-23","arxiv_id":"2205.11019","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-in-the-loop-provably-efficient","title":"Human-in-the-loop: Provably Efficient Preference-based Reinforcement Learning with General Function Approximation","date":"2022-05-23","arxiv_id":"2205.11140","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-advise-and-learning-from-advice","title":"Learning to Advise and Learning from Advice in Cooperative Multi-Agent Reinforcement Learning","date":"2022-05-23","arxiv_id":"2205.11163","repositories_listed":0,"syntology":null},{"url":null,"slug":"logarithmic-regret-bounds-for-continuous-time","title":"Logarithmic regret bounds for continuous-time average-reward Markov decision processes","date":"2022-05-23","arxiv_id":"2205.11168","repositories_listed":0,"syntology":null},{"url":null,"slug":"multiple-domain-cyberspace-attack-and-defense","title":"Multiple Domain Cyberspace Attack and Defense Game Based on Reward Randomization Reinforcement Learning","date":"2022-05-23","arxiv_id":"2205.10990","repositories_listed":0,"syntology":null},{"url":null,"slug":"polter-policy-trajectory-ensemble","title":"POLTER: Policy Trajectory Ensemble Regularization for Unsupervised Reinforcement Learning","date":"2022-05-23","arxiv_id":"2205.11357","repositories_listed":0,"syntology":null},{"url":null,"slug":"rl-with-kl-penalties-is-better-viewed-as","title":"RL with KL penalties is better viewed as Bayesian inference","date":"2022-05-23","arxiv_id":"2205.11275","repositories_listed":0,"syntology":null},{"url":null,"slug":"spreading-factor-and-rssi-for-localization-in","title":"Spreading Factor assisted LoRa Localization with Deep Reinforcement Learning","date":"2022-05-23","arxiv_id":"2205.11428","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-dirichlet-process-mixture-of-robust-task","title":"A Dirichlet Process Mixture of Robust Task Models for Scalable Lifelong Reinforcement Learning","date":"2022-05-22","arxiv_id":"2205.10787","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-information-directed-sampling","title":"Contextual Information-Directed Sampling","date":"2022-05-22","arxiv_id":"2205.10895","repositories_listed":0,"syntology":null},{"url":null,"slug":"inverse-inverse-reinforcement-learning-how-to","title":"Inverse-Inverse Reinforcement Learning. How to Hide Strategy from an Adversarial Inverse Reinforcement Learner","date":"2022-05-22","arxiv_id":"2205.10802","repositories_listed":0,"syntology":null},{"url":null,"slug":"power-and-accountability-in-reinforcement","title":"Power and accountability in reinforcement learning applications to environmental policy","date":"2022-05-22","arxiv_id":"2205.10911","repositories_listed":0,"syntology":null},{"url":null,"slug":"coordinating-policies-among-multiple-agents","title":"Coordinating Policies Among Multiple Agents via an Intelligent Communication Channel","date":"2022-05-21","arxiv_id":"2205.10607","repositories_listed":0,"syntology":null},{"url":null,"slug":"coral-contextual-response-retrievability-loss","title":"CORAL: Contextual Response Retrievability Loss Function for Training Dialog Generation Models","date":"2022-05-21","arxiv_id":"2205.10558","repositories_listed":0,"syntology":null},{"url":null,"slug":"de-novo-design-of-protein-target-specific-1","title":"De novo design of protein target specific scaffold-based Inhibitors via Reinforcement Learning","date":"2022-05-21","arxiv_id":"2205.10473","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforced-pedestrian-attribute-recognition","title":"Reinforced Pedestrian Attribute Recognition with Group Optimization Reward","date":"2022-05-21","arxiv_id":"2205.14042","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-fully-controllable-agent-in-the-path","title":"A Fully Controllable Agent in the Path Planning using Goal-Conditioned Reinforcement Learning","date":"2022-05-20","arxiv_id":"2205.09967","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-joint-attacks-on-legged-robots","title":"Adversarial joint attacks on legged robots","date":"2022-05-20","arxiv_id":"2205.10098","repositories_listed":0,"syntology":null},{"url":null,"slug":"long-run-incremental-cost-lric-distribution","title":"Long Run Incremental Cost (LRIC) Distribution Network Pricing in UK, advising China's Distribution Network","date":"2022-05-20","arxiv_id":"2205.09946","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-jointly-optimizing-partial-offloading-and","title":"On Jointly Optimizing Partial Offloading and SFC Mapping: A Cooperative Dual-agent Deep Reinforcement Learning Approach","date":"2022-05-20","arxiv_id":"2205.09925","repositories_listed":0,"syntology":null},{"url":null,"slug":"prototyping-three-key-properties-of-specific","title":"Prototyping three key properties of specific curiosity in computational reinforcement learning","date":"2022-05-20","arxiv_id":"2205.10407","repositories_listed":0,"syntology":null},{"url":null,"slug":"survey-on-fair-reinforcement-learning-theory","title":"Survey on Fair Reinforcement Learning: Theory and Practice","date":"2022-05-20","arxiv_id":"2205.10032","repositories_listed":0,"syntology":null},{"url":null,"slug":"aigenc-ai-generalisation-via-creativity","title":"AIGenC: An AI generalisation model via creativity","date":"2022-05-19","arxiv_id":"2205.09738","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-valuation-for-offline-reinforcement","title":"Data Valuation for Offline Reinforcement Learning","date":"2022-05-19","arxiv_id":"2205.09550","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-multi-agent-deep-reinforcement-1","title":"Distributed Multi-Agent Deep Reinforcement Learning for Robust Coordination against Noise","date":"2022-05-19","arxiv_id":"2205.09705","repositories_listed":0,"syntology":null},{"url":null,"slug":"il-flow-imitation-learning-from-observation","title":"IL-flOw: Imitation Learning from Observation using Normalizing Flows","date":"2022-05-19","arxiv_id":"2205.09251","repositories_listed":0,"syntology":null},{"url":null,"slug":"parallel-bandit-architecture-based-on-laser","title":"Parallel bandit architecture based on laser chaos for reinforcement learning","date":"2022-05-19","arxiv_id":"2205.09543","repositories_listed":0,"syntology":null},{"url":null,"slug":"routing-and-placement-of-macros-using-deep","title":"Routing and Placement of Macros using Deep Reinforcement Learning","date":"2022-05-19","arxiv_id":"2205.09289","repositories_listed":0,"syntology":null},{"url":null,"slug":"sparse-adversarial-attack-in-multi-agent","title":"Sparse Adversarial Attack in Multi-agent Reinforcement Learning","date":"2022-05-19","arxiv_id":"2205.09362","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-applicable-reinforcement-learning","title":"Towards Applicable Reinforcement Learning: Improving the Generalization and Sample Efficiency with Policy Ensemble","date":"2022-05-19","arxiv_id":"2205.09284","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-on-location","title":"Deep Reinforcement Learning Based on Location-Aware Imitation Environment for RIS-Aided mmWave MIMO Systems","date":"2022-05-18","arxiv_id":"2205.08788","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-explanations-from-deep","title":"Generating Explanations from Deep Reinforcement Learning Using Episodic Memory","date":"2022-05-18","arxiv_id":"2205.08926","repositories_listed":0,"syntology":null},{"url":null,"slug":"market-making-via-reinforcement-learning-in","title":"Market Making via Reinforcement Learning in China Commodity Market","date":"2022-05-18","arxiv_id":"2205.08936","repositories_listed":0,"syntology":null},{"url":null,"slug":"no-more-pesky-hyperparameters-offline","title":"No More Pesky Hyperparameters: Offline Hyperparameter Tuning for RL","date":"2022-05-18","arxiv_id":"2205.08716","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-distillation-with-selective-input","title":"Policy Distillation with Selective Input Gradient Regularization for Efficient Interpretability","date":"2022-05-18","arxiv_id":"2205.08685","repositories_listed":0,"syntology":null},{"url":null,"slug":"slowly-changing-adversarial-bandit-algorithms","title":"Slowly Changing Adversarial Bandit Algorithms are Efficient for Discounted MDPs","date":"2022-05-18","arxiv_id":"2205.09056","repositories_listed":0,"syntology":null},{"url":null,"slug":"world-value-functions-knowledge","title":"World Value Functions: Knowledge Representation for Multitask Reinforcement Learning","date":"2022-05-18","arxiv_id":"2205.08827","repositories_listed":0,"syntology":null},{"url":null,"slug":"moral-reinforcement-learning-using-actual","title":"Moral reinforcement learning using actual causation","date":"2022-05-17","arxiv_id":"2205.08192","repositories_listed":0,"syntology":null},{"url":null,"slug":"multibit-tries-packet-classification-with","title":"Multibit Tries Packet Classification with Deep Reinforcement Learning","date":"2022-05-17","arxiv_id":"2205.08606","repositories_listed":0,"syntology":null},{"url":null,"slug":"planning-to-practice-efficient-online-fine","title":"Planning to Practice: Efficient Online Fine-Tuning by Composing Goals in Latent Space","date":"2022-05-17","arxiv_id":"2205.08129","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-blind-ai-in","title":"A Deep Reinforcement Learning Blind AI in DareFightingICE","date":"2022-05-16","arxiv_id":"2205.07444","repositories_listed":0,"syntology":null},{"url":null,"slug":"attacking-and-defending-deep-reinforcement","title":"Attacking and Defending Deep Reinforcement Learning Policies","date":"2022-05-16","arxiv_id":"2205.07626","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-apprenticeship-learning-for-playing","title":"Deep Apprenticeship Learning for Playing Games","date":"2022-05-16","arxiv_id":"2205.07959","repositories_listed":0,"syntology":null},{"url":null,"slug":"enforcing-kl-regularization-in-general","title":"Enforcing KL Regularization in General Tsallis Entropy Reinforcement Learning via Advantage Learning","date":"2022-05-16","arxiv_id":"2205.07885","repositories_listed":0,"syntology":null},{"url":null,"slug":"kgrgrl-a-user-s-permission-reasoning-method","title":"KGRGRL: A User's Permission Reasoning Method Based on Knowledge Graph Reward Guidance Reinforcement Learning","date":"2022-05-16","arxiv_id":"2205.07502","repositories_listed":0,"syntology":null},{"url":null,"slug":"many-field-packet-classification-with","title":"Many Field Packet Classification with Decomposition and Reinforcement Learning","date":"2022-05-16","arxiv_id":"2205.07973","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-munchausen-reinforcement-learning","title":"$q$-Munchausen Reinforcement Learning","date":"2022-05-16","arxiv_id":"2205.07467","repositories_listed":0,"syntology":null},{"url":null,"slug":"qualitative-differences-between-evolutionary","title":"Qualitative Differences Between Evolutionary Strategies and Reinforcement Learning Methods for Control of Autonomous Agents","date":"2022-05-16","arxiv_id":"2205.07592","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-reinforcement-learning-based-logic","title":"Rethinking Reinforcement Learning based Logic Synthesis","date":"2022-05-16","arxiv_id":"2205.07614","repositories_listed":0,"syntology":null},{"url":null,"slug":"taming-continuous-posteriors-for-latent","title":"Taming Continuous Posteriors for Latent Variational Dialogue Policies","date":"2022-05-16","arxiv_id":"2205.07633","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-on-sky-adaptive-optics-control-using","title":"Towards on-sky adaptive optics control using reinforcement learning","date":"2022-05-16","arxiv_id":"2205.07554","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-gradient-method-for-robust","title":"Policy Gradient Method For Robust Reinforcement Learning","date":"2022-05-15","arxiv_id":"2205.07344","repositories_listed":0,"syntology":null},{"url":null,"slug":"romfac-a-robust-mean-field-actor-critic","title":"RoMFAC: A robust mean-field actor-critic reinforcement learning against adversarial perturbations on states","date":"2022-05-15","arxiv_id":"2205.07229","repositories_listed":0,"syntology":null},{"url":"/paper/cliff-diving-exploring-reward-surfaces-in","slug":"cliff-diving-exploring-reward-surfaces-in","title":"Cliff Diving: Exploring Reward Surfaces in Reinforcement Learning Environments","date":"2022-05-14","arxiv_id":"2205.07015","repositories_listed":0,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/cliff-diving-exploring-reward-surfaces-in#ran","syntology_url":"https://syntology.ai/paper/2205.07015","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.07015"}},"official":null}},{"url":null,"slug":"interpretable-stochastic-model-predictive","title":"Interpretable Stochastic Model Predictive Control using Distributional Reinforced Estimation for Quadrotor Tracking Systems","date":"2022-05-14","arxiv_id":"2205.07150","repositories_listed":0,"syntology":null},{"url":null,"slug":"prefixrl-optimization-of-parallel-prefix","title":"PrefixRL: Optimization of Parallel Prefix Circuits using Deep Reinforcement Learning","date":"2022-05-14","arxiv_id":"2205.07000","repositories_listed":0,"syntology":null},{"url":null,"slug":"qhd-a-brain-inspired-hyperdimensional","title":"Efficient Off-Policy Reinforcement Learning via Brain-Inspired Computing","date":"2022-05-14","arxiv_id":"2205.06978","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-in-mmw-noma-joint","title":"Deep Reinforcement Learning in mmW-NOMA: Joint Power Allocation and Hybrid Beamforming","date":"2022-05-13","arxiv_id":"2205.06814","repositories_listed":0,"syntology":null},{"url":null,"slug":"emergent-bartering-behaviour-in-multi-agent","title":"Emergent Bartering Behaviour in Multi-Agent Reinforcement Learning","date":"2022-05-13","arxiv_id":"2205.06760","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-power-allocation-and-beamformer-for-mmw","title":"Joint Power Allocation and Beamformer for mmW-NOMA Downlink Systems by Deep Reinforcement Learning","date":"2022-05-13","arxiv_id":"2205.06489","repositories_listed":0,"syntology":null}],"record_sha256":"b72fa43f8a9e9b16a6281902a68e952063e52fb085f22f11c1c2c6a320f841cf","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}