{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/124","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":124,"pages_in_order":135,"rows_per_page":100,"rows":[12301,12400],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/123","next":"/task/reinforcement-learning-2/papers/125","papers":[{"url":null,"slug":"integrating-reinforcement-learning-to-self","title":"Integrating Reinforcement Learning to Self Training for Pulmonary Nodule Segmentation in Chest X-rays","date":"2018-11-21","arxiv_id":"1811.08840","repositories_listed":0,"syntology":null},{"url":null,"slug":"energy-efficiency-in-reinforcement-learning","title":"Energy Efficiency in Reinforcement Learning for Wireless Sensor Networks","date":"2018-11-19","arxiv_id":"1812.02538","repositories_listed":0,"syntology":null},{"url":null,"slug":"measurement-based-adaptation-protocol-with","title":"Measurement-based adaptation protocol with quantum reinforcement learning in a Rigetti quantum computer","date":"2018-11-19","arxiv_id":"1811.07594","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-and-inverse","title":"Reinforcement Learning and Inverse Reinforcement Learning with System 1 and System 2","date":"2018-11-19","arxiv_id":"1811.08549","repositories_listed":0,"syntology":null},{"url":null,"slug":"simulated-autonomous-driving-in-a-realistic","title":"Simulated Autonomous Driving in a Realistic Driving Environment using Deep Reinforcement Learning and a Deterministic Finite State Machine","date":"2018-11-19","arxiv_id":"1811.07868","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-optimization-with-model-based","title":"Policy Optimization with Model-based Explorations","date":"2018-11-18","arxiv_id":"1811.07350","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-organizing-maps-for-storage-and-transfer","title":"Self-Organizing Maps for Storage and Transfer of Knowledge in Reinforcement Learning","date":"2018-11-18","arxiv_id":"1811.08318","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-extraction-of-a-hierarchical","title":"Autonomous Extraction of a Hierarchical Structure of Tasks in Reinforcement Learning, A Sequential Associate Rule Mining Approach","date":"2018-11-17","arxiv_id":"1811.08275","repositories_listed":0,"syntology":null},{"url":null,"slug":"emergence-of-linguistic-conventions-in-multi","title":"Emergence of linguistic conventions in multi-agent reinforcement learning","date":"2018-11-17","arxiv_id":"1811.07208","repositories_listed":0,"syntology":null},{"url":null,"slug":"parameter-sharing-reinforcement-learning","title":"Parameter Sharing Reinforcement Learning Architecture for Multi Agent Driving Behaviors","date":"2018-11-17","arxiv_id":"1811.07214","repositories_listed":0,"syntology":null},{"url":null,"slug":"recursive-sparse-pseudo-input-gaussian","title":"Recursive Sparse Pseudo-input Gaussian Process SARSA","date":"2018-11-17","arxiv_id":"1811.07201","repositories_listed":0,"syntology":null},{"url":null,"slug":"concept-learning-through-deep-reinforcement","title":"Concept Learning through Deep Reinforcement Learning with Memory-Augmented Neural Networks","date":"2018-11-15","arxiv_id":"1811.06145","repositories_listed":0,"syntology":null},{"url":null,"slug":"intervention-aided-reinforcement-learning-for","title":"Intervention Aided Reinforcement Learning for Safe and Practical Policy Optimization in Navigation","date":"2018-11-15","arxiv_id":"1811.06187","repositories_listed":0,"syntology":null},{"url":null,"slug":"orthogonal-policy-gradient-and-autonomous","title":"Orthogonal Policy Gradient and Autonomous Driving Application","date":"2018-11-15","arxiv_id":"1811.06151","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-utility-of-sparse-representations-for","title":"The Utility of Sparse Representations for Control in Reinforcement Learning","date":"2018-11-15","arxiv_id":"1811.06626","repositories_listed":0,"syntology":null},{"url":null,"slug":"bayesian-reinforcement-learning-in-factored","title":"Bayesian Reinforcement Learning in Factored POMDPs","date":"2018-11-14","arxiv_id":"1811.05612","repositories_listed":0,"syntology":null},{"url":null,"slug":"emergence-of-addictive-behaviors-in","title":"Emergence of Addictive Behaviors in Reinforcement Learning Agents","date":"2018-11-14","arxiv_id":"1811.05590","repositories_listed":0,"syntology":null},{"url":null,"slug":"coordinating-disaster-emergency-response-with","title":"Coordinating Disaster Emergency Response with Heuristic Reinforcement Learning","date":"2018-11-12","arxiv_id":"1811.05010","repositories_listed":0,"syntology":null},{"url":null,"slug":"importance-weighted-evolution-strategies","title":"Importance Weighted Evolution Strategies","date":"2018-11-12","arxiv_id":"1811.04624","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-data-augmentation-policies-using","title":"Learning data augmentation policies using augmented random search","date":"2018-11-12","arxiv_id":"1811.04768","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-temporal-point-processes-via","title":"Learning Temporal Point Processes via Reinforcement Learning","date":"2018-11-12","arxiv_id":"1811.05016","repositories_listed":0,"syntology":null},{"url":null,"slug":"navigating-assistance-system-for-quadcopter","title":"Navigating Assistance System for Quadcopter with Deep Reinforcement Learning","date":"2018-11-12","arxiv_id":"1811.04584","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-initial-attempt-of-combining-visual","title":"An initial attempt of combining visual selective attention with deep reinforcement learning","date":"2018-11-11","arxiv_id":"1811.04407","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-optimal-control-view-of-adversarial","title":"An Optimal Control View of Adversarial Machine Learning","date":"2018-11-11","arxiv_id":"1811.04422","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-taxi-carpool-policies-via","title":"Optimizing Taxi Carpool Policies via Reinforcement Learning and Spatio-Temporal Mining","date":"2018-11-11","arxiv_id":"1811.04345","repositories_listed":0,"syntology":null},{"url":null,"slug":"product-title-refinement-via-multi-modal","title":"Product Title Refinement via Multi-Modal Generative Adversarial Learning","date":"2018-11-11","arxiv_id":"1811.04498","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-governing-agents-efficacy-action","title":"Towards Governing Agent's Efficacy: Action-Conditional $β$-VAE for Deep Transparent Reinforcement Learning","date":"2018-11-11","arxiv_id":"1811.04350","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-shaping-strategies-in-human-in-the","title":"Learning Shaping Strategies in Human-in-the-loop Interactive Reinforcement Learning","date":"2018-11-10","arxiv_id":"1811.04272","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-speech","title":"Reinforcement Learning Based Speech Enhancement for Robust Speech Recognition","date":"2018-11-10","arxiv_id":"1811.04224","repositories_listed":0,"syntology":null},{"url":null,"slug":"correlation-filter-selection-for-visual","title":"Correlation Filter Selection for Visual Tracking Using Reinforcement Learning","date":"2018-11-08","arxiv_id":"1811.03196","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-learning-for-multi-objective","title":"Meta-Learning for Multi-objective Reinforcement Learning","date":"2018-11-08","arxiv_id":"1811.03376","repositories_listed":0,"syntology":null},{"url":null,"slug":"modular-architecture-for-starcraft-ii-with","title":"Modular Architecture for StarCraft II with Deep Reinforcement Learning","date":"2018-11-08","arxiv_id":"1811.03555","repositories_listed":0,"syntology":null},{"url":null,"slug":"baselines-for-reinforcement-learning-in-text","title":"Baselines for Reinforcement Learning in Text Games","date":"2018-11-07","arxiv_id":"1811.02872","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-certificates-towards-accountable","title":"Policy Certificates: Towards Accountable Reinforcement Learning","date":"2018-11-07","arxiv_id":"1811.03056","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-stress-testing-finding-failure","title":"Adaptive Stress Testing: Finding Likely Failure Events with Reinforcement Learning","date":"2018-11-06","arxiv_id":"1811.02188","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-green","title":"Deep Reinforcement Learning for Green Security Games with Real-Time Information","date":"2018-11-06","arxiv_id":"1811.02483","repositories_listed":0,"syntology":null},{"url":null,"slug":"quasi-newton-optimization-in-deep-q-learning","title":"Deep Reinforcement Learning via L-BFGS Optimization","date":"2018-11-06","arxiv_id":"1811.02693","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-continual-learning-in-medical-imaging","title":"Towards continual learning in medical imaging","date":"2018-11-06","arxiv_id":"1811.02496","repositories_listed":0,"syntology":null},{"url":null,"slug":"combining-subgoal-graphs-with-reinforcement","title":"Combining Subgoal Graphs with Reinforcement Learning to Build a Rational Pathfinder","date":"2018-11-05","arxiv_id":"1811.01700","repositories_listed":0,"syntology":null},{"url":"/paper/contingency-aware-exploration-in","slug":"contingency-aware-exploration-in","title":"Contingency-Aware Exploration in Reinforcement Learning","date":"2018-11-05","arxiv_id":"1811.01483","repositories_listed":0,"syntology":null},{"url":null,"slug":"managing-engineering-systems-with-large-state","title":"Managing engineering systems with large state and action spaces through deep reinforcement learning","date":"2018-11-05","arxiv_id":"1811.02052","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-dynamic-model","title":"Reinforcement Learning based Dynamic Model Selection for Short-Term Load Forecasting","date":"2018-11-05","arxiv_id":"1811.01846","repositories_listed":0,"syntology":null},{"url":null,"slug":"releq-an-automatic-reinforcement-learning","title":"ReLeQ: A Reinforcement Learning Approach for Deep Quantization of Neural Networks","date":"2018-11-05","arxiv_id":"1811.01704","repositories_listed":0,"syntology":null},{"url":null,"slug":"relation-mention-extraction-from-noisy-data","title":"Relation Mention Extraction from Noisy Data with Hierarchical Reinforcement Learning","date":"2018-11-03","arxiv_id":"1811.01237","repositories_listed":0,"syntology":null},{"url":null,"slug":"automated-theorem-proving-in-intuitionistic","title":"Automated Theorem Proving in Intuitionistic Propositional Logic by Deep Reinforcement Learning","date":"2018-11-02","arxiv_id":"1811.00796","repositories_listed":0,"syntology":null},{"url":null,"slug":"approximate-dynamic-oracle-for-dependency","title":"Approximate Dynamic Oracle for Dependency Parsing with Reinforcement Learning","date":"2018-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-modeling-for-query-expansion-and","title":"Joint Modeling for Query Expansion and Information Extraction with Reinforcement Learning","date":"2018-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"macquarie-university-at-bioasq-6b-deep","title":"Macquarie University at BioASQ 6b: Deep learning and deep reinforcement learning for query-based summarisation","date":"2018-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"shaping-a-social-robots-humor-with-natural","title":"Shaping a social robot's humor with Natural Language Generation and socially-aware reinforcement learning","date":"2018-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"sdrl-interpretable-and-data-efficient-deep","title":"SDRL: Interpretable and Data-efficient Deep Reinforcement Learning Leveraging Symbolic Planning","date":"2018-10-31","arxiv_id":"1811.00090","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-a-simple-approach-to-multi-step-model","title":"Towards a Simple Approach to Multi-step Model-based Reinforcement Learning","date":"2018-10-31","arxiv_id":"1811.00128","repositories_listed":0,"syntology":null},{"url":null,"slug":"relative-importance-sampling-for-off-policy","title":"Relative Importance Sampling for off-Policy Actor-Critic in Deep Reinforcement Learning","date":"2018-10-30","arxiv_id":"1810.12558","repositories_listed":0,"syntology":null},{"url":null,"slug":"social-vehicle-swarms-a-novel-perspective-on","title":"Social Vehicle Swarms: A Novel Perspective on Social-aware Vehicular Communication Architecture","date":"2018-10-29","arxiv_id":"1810.11947","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributive-dynamic-spectrum-access-through","title":"Distributive Dynamic Spectrum Access through Deep Reinforcement Learning: A Reservoir Computing Based Approach","date":"2018-10-28","arxiv_id":"1810.11758","repositories_listed":0,"syntology":null},{"url":null,"slug":"empirical-evaluation-of-contextual-policy","title":"Empirical Evaluation of Contextual Policy Search with a Comparison-based Surrogate Model and Active Covariance Matrix Adaptation","date":"2018-10-26","arxiv_id":"1810.11491","repositories_listed":0,"syntology":null},{"url":null,"slug":"stability-certified-reinforcement-learning-a","title":"Stability-certified reinforcement learning: A control-theoretic perspective","date":"2018-10-26","arxiv_id":"1810.11505","repositories_listed":0,"syntology":null},{"url":null,"slug":"differential-variable-speed-limits-control","title":"Differential Variable Speed Limits Control for Freeway Recurrent Bottlenecks via Deep Reinforcement learning","date":"2018-10-25","arxiv_id":"1810.10952","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-based-1","title":"Multi-Agent Reinforcement Learning Based Resource Allocation for UAV Networks","date":"2018-10-24","arxiv_id":"1810.10408","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-approaches-for-reinforcement","title":"Hierarchical Approaches for Reinforcement Learning in Parameterized Action Space","date":"2018-10-23","arxiv_id":"1810.09656","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-representations-in-model-free","title":"Learning Representations in Model-Free Hierarchical Reinforcement Learning","date":"2018-10-23","arxiv_id":"1810.10096","repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-sensitive-reinforcement-learning-a","title":"Risk-Sensitive Reinforcement Learning via Policy Gradient Search","date":"2018-10-22","arxiv_id":"1810.09126","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-self-explanation-of-behavior-for","title":"Autonomous Self-Explanation of Behavior for Interactive Reinforcement Learning Agents","date":"2018-10-20","arxiv_id":"1810.08811","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-with-model","title":"Safe Reinforcement Learning with Model Uncertainty Estimates","date":"2018-10-19","arxiv_id":"1810.08700","repositories_listed":0,"syntology":null},{"url":null,"slug":"applications-of-deep-reinforcement-learning","title":"Applications of Deep Reinforcement Learning in Communications and Networking: A Survey","date":"2018-10-18","arxiv_id":"1810.07862","repositories_listed":0,"syntology":null},{"url":null,"slug":"finding-the-best-design-parameters-for","title":"Finding the best design parameters for optical nanostructures using reinforcement learning","date":"2018-10-18","arxiv_id":"1810.10964","repositories_listed":0,"syntology":null},{"url":null,"slug":"at-human-speed-deep-reinforcement-learning","title":"At Human Speed: Deep Reinforcement Learning with Action Delay","date":"2018-10-16","arxiv_id":"1810.07286","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-concept-of-criticality-in-reinforcement","title":"The Concept of Criticality in Reinforcement Learning","date":"2018-10-16","arxiv_id":"1810.07254","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-deep-reinforcement-learning-for-the","title":"Using Deep Reinforcement Learning for the Continuous Control of Robotic Arms","date":"2018-10-15","arxiv_id":"1810.06746","repositories_listed":0,"syntology":null},{"url":null,"slug":"dexterous-manipulation-with-deep","title":"Dexterous Manipulation with Deep Reinforcement Learning: Efficient, General, and Low-Cost","date":"2018-10-14","arxiv_id":"1810.06045","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-multiagent-deep-reinforcement-learning-the","title":"A Survey and Critique of Multiagent Deep Reinforcement Learning","date":"2018-10-12","arxiv_id":"1810.05587","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-hierarchical-learning-path-design","title":"Optimal Hierarchical Learning Path Design with Reinforcement Learning","date":"2018-10-12","arxiv_id":"1810.05347","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-text-generation-without","title":"Adversarial Text Generation Without Reinforcement Learning","date":"2018-10-11","arxiv_id":"1810.06640","repositories_listed":0,"syntology":null},{"url":null,"slug":"continual-state-representation-learning-for","title":"Continual State Representation Learning for Reinforcement Learning using Generative Replay","date":"2018-10-09","arxiv_id":"1810.03880","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-wildfire-surveillance-with","title":"Distributed Wildfire Surveillance with Autonomous Aircraft using Deep Reinforcement Learning","date":"2018-10-09","arxiv_id":"1810.04244","repositories_listed":0,"syntology":null},{"url":null,"slug":"enabling-cognitive-smart-cities-using-big","title":"Enabling Cognitive Smart Cities Using Big Data and Machine Learning: Approaches and Challenges","date":"2018-10-09","arxiv_id":"1810.04107","repositories_listed":0,"syntology":null},{"url":null,"slug":"actor-critic-deep-reinforcement-learning-for","title":"Actor-Critic Deep Reinforcement Learning for Dynamic Multichannel Access","date":"2018-10-08","arxiv_id":"1810.03695","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-deep-reinforcement-learning-for","title":"Multi-agent Deep Reinforcement Learning for Zero Energy Communities","date":"2018-10-08","arxiv_id":"1810.03679","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-evolutionary-learning-method","title":"Reinforcement Evolutionary Learning Method for self-learning","date":"2018-10-07","arxiv_id":"1810.03198","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-time","title":"Deep Reinforcement Learning for Time Scheduling in RF-Powered Backscatter Cognitive Radio Networks","date":"2018-10-03","arxiv_id":"1810.04520","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-essay-scoring-incorporating-rating","title":"Automatic Essay Scoring Incorporating Rating Schema via Reinforcement Learning","date":"2018-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-poetry-generation-with-mutual","title":"Automatic Poetry Generation with Mutual Reinforcement Learning","date":"2018-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-sub-domain-modeling-for-dialogue","title":"Autonomous Sub-domain Modeling for Dialogue Policy with Hierarchical Deep Reinforcement Learning","date":"2018-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"curriculum-learning-based-on-reward","title":"Curriculum Learning Based on Reward Sparseness for Deep Reinforcement Learning of Task Completion Dialogue Management","date":"2018-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"logician-and-orator-learning-from-the-duality","title":"Logician and Orator: Learning from the Duality between Language and Knowledge in Open Domain","date":"2018-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"prediction-improves-simultaneous-neural","title":"Prediction Improves Simultaneous Neural Machine Translation","date":"2018-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"bayesian-transfer-reinforcement-learning-with","title":"Bayesian Transfer Reinforcement Learning with Prior Knowledge Rules","date":"2018-09-30","arxiv_id":"1810.00468","repositories_listed":0,"syntology":null},{"url":null,"slug":"few-shot-goal-inference-for-visuomotor","title":"Few-Shot Goal Inference for Visuomotor Learning and Planning","date":"2018-09-30","arxiv_id":"1810.00482","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-r","title":"Reinforcement Learning in R","date":"2018-09-29","arxiv_id":"1810.00240","repositories_listed":0,"syntology":null},{"url":null,"slug":"direct-optimization-of-f-measure-for","title":"Direct optimization of F-measure for retrieval-based personal question answering","date":"2018-09-28","arxiv_id":"1810.00679","repositories_listed":0,"syntology":null},{"url":null,"slug":"robot-representation-and-reasoning-with","title":"Robot Representation and Reasoning with Knowledge from Reinforcement Learning","date":"2018-09-28","arxiv_id":"1809.11074","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-better-baseline-for-second-order-gradient","title":"A Better Baseline for Second Order Gradient Estimation in Stochastic Computation Graphs","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-convergent-variant-of-the-boltzmann-softmax","title":"A Convergent Variant of the Boltzmann Softmax Operator in Reinforcement Learning","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerated-value-iteration-via-anderson","title":"Accelerated Value Iteration via Anderson Mixing","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"collaborative-multiagent-reinforcement","title":"COLLABORATIVE MULTIAGENT REINFORCEMENT LEARNING IN HOMOGENEOUS SWARMS","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"constraining-action-sequences-with-formal","title":"Constraining Action Sequences with Formal Languages for Deep Reinforcement Learning","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"convergent-reinforcement-learning-with","title":"Convergent Reinforcement Learning with Function Approximation: A Bilevel Optimization Perspective","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-of-universal","title":"Deep Reinforcement Learning of Universal Policies with Diverse Environment Summaries","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"definition-and-evaluation-of-model-free","title":"Definition and evaluation of model-free coordination of electrical vehicle charging with reinforcement learning","date":"2018-09-27","arxiv_id":"1809.10679","repositories_listed":0,"syntology":null},{"url":null,"slug":"distilled-agent-dqn-for-provable-adversarial","title":"Distilled Agent DQN for Provable Adversarial Robustness","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-pricing-on-e-commerce-platform-with-1","title":"Dynamic Pricing on E-commerce Platform with Deep Reinforcement Learning","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null}],"record_sha256":"b59cbdd2a13c951d59551967e24bfbe46a840a355e9cd358afba6e6529f7e27c","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}