{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/120","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":120,"pages_in_order":135,"rows_per_page":100,"rows":[11901,12000],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/119","next":"/task/reinforcement-learning-2/papers/121","papers":[{"url":null,"slug":"linear-interpolation-gives-better-gradients","title":"Linear interpolation gives better gradients than Gaussian smoothing in derivative-free optimization","date":"2019-05-29","arxiv_id":"1905.13043","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-generalization-gap-in","title":"On the Generalization Gap in Reparameterizable Reinforcement Learning","date":"2019-05-29","arxiv_id":"1905.12654","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-policy-mixture","title":"Reinforcement Learning with Policy Mixture Model for Temporal Point Processes Clustering","date":"2019-05-29","arxiv_id":"1905.12345","repositories_listed":0,"syntology":null},{"url":null,"slug":"switching-linear-dynamics-for-variational","title":"Switching Linear Dynamics for Variational Bayes Filtering","date":"2019-05-29","arxiv_id":"1905.12434","repositories_listed":0,"syntology":null},{"url":null,"slug":"targeted-attacks-on-deep-reinforcement","title":"CopyCAT: Taking Control of Neural Policies with Constant Attacks","date":"2019-05-29","arxiv_id":"1905.12282","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-exponentially-discounted-sum-automatic","title":"Beyond Exponentially Discounted Sum: Automatic Learning of Return Function","date":"2019-05-28","arxiv_id":"1905.11591","repositories_listed":0,"syntology":null},{"url":null,"slug":"conditions-on-features-for-temporal","title":"Conditions on Features for Temporal Difference-Like Methods to Converge","date":"2019-05-28","arxiv_id":"1905.11702","repositories_listed":0,"syntology":null},{"url":null,"slug":"generation-of-policy-level-explanations-for","title":"Generation of Policy-Level Explanations for Reinforcement Learning","date":"2019-05-28","arxiv_id":"1905.12044","repositories_listed":0,"syntology":null},{"url":null,"slug":"interactive-teaching-algorithms-for-inverse","title":"Interactive Teaching Algorithms for Inverse Reinforcement Learning","date":"2019-05-28","arxiv_id":"1905.11867","repositories_listed":0,"syntology":null},{"url":null,"slug":"190601408","title":"Hypothesis-Driven Skill Discovery for Hierarchical Deep Reinforcement Learning","date":"2019-05-27","arxiv_id":"1906.01408","repositories_listed":0,"syntology":null},{"url":null,"slug":"agentgraph-towards-universal-dialogue","title":"AgentGraph: Towards Universal Dialogue Management with Structured Deep Reinforcement Learning","date":"2019-05-27","arxiv_id":"1905.11259","repositories_listed":0,"syntology":null},{"url":null,"slug":"near-optimal-optimistic-reinforcement","title":"Near-optimal Optimistic Reinforcement Learning using Empirical Bernstein Inequalities","date":"2019-05-27","arxiv_id":"1905.12425","repositories_listed":0,"syntology":null},{"url":null,"slug":"variational-bayes-a-report-on-approaches-and","title":"Variational Bayes: A report on approaches and applications","date":"2019-05-26","arxiv_id":"1905.10744","repositories_listed":0,"syntology":null},{"url":null,"slug":"190512726","title":"Prioritized Sequence Experience Replay","date":"2019-05-25","arxiv_id":"1905.12726","repositories_listed":0,"syntology":null},{"url":null,"slug":"aspire-automated-security-policy","title":"Transferable Cost-Aware Security Policy Implementation for Malware Detection Using Deep Reinforcement Learning","date":"2019-05-25","arxiv_id":"1905.10517","repositories_listed":0,"syntology":null},{"url":null,"slug":"composing-ensembles-of-policies-with-deep","title":"Composing Task-Agnostic Policies with Deep Reinforcement Learning","date":"2019-05-25","arxiv_id":"1905.10681","repositories_listed":0,"syntology":null},{"url":"/paper/learning-to-reason-in-large-theories-without","slug":"learning-to-reason-in-large-theories-without","title":"Learning to Reason in Large Theories without Imitation","date":"2019-05-25","arxiv_id":"1905.10501","repositories_listed":0,"syntology":null},{"url":null,"slug":"190512567","title":"MQLV: Optimal Policy of Money Management in Retail Banking with Q-Learning","date":"2019-05-24","arxiv_id":"1905.12567","repositories_listed":0,"syntology":null},{"url":null,"slug":"190601407","title":"RL4health: Crowdsourcing Reinforcement Learning for Knee Replacement Pathway Optimization","date":"2019-05-24","arxiv_id":"1906.01407","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-machine-learning-by-pipeline","title":"Automatic Machine Learning by Pipeline Synthesis using Model-Based Reinforcement Learning and a Grammar","date":"2019-05-24","arxiv_id":"1905.10345","repositories_listed":0,"syntology":null},{"url":null,"slug":"inforl-interpretable-reinforcement-learning","title":"InfoRL: Interpretable Reinforcement Learning using Information Maximization","date":"2019-05-24","arxiv_id":"1905.10404","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-leaning-in-feature-space-matrix","title":"Reinforcement Learning in Feature Space: Matrix Bandit, Kernels, and Regret Bound","date":"2019-05-24","arxiv_id":"1905.10389","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-expected-cumulative-reward","title":"A Micro-Objective Perspective of Reinforcement Learning","date":"2019-05-24","arxiv_id":"1905.10016","repositories_listed":0,"syntology":null},{"url":null,"slug":"190509949","title":"Scene Induced Multi-Modal Trajectory Forecasting via Planning","date":"2019-05-23","arxiv_id":"1905.09949","repositories_listed":0,"syntology":null},{"url":null,"slug":"average-reward-reinforcement-learning-with","title":"Unknown mixing times in apprenticeship and reinforcement learning","date":"2019-05-23","arxiv_id":"1905.09704","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-q-learning-with-q-matrix-transfer","title":"Deep Q-Learning with Q-Matrix Transfer Learning for Novel Fire Evacuation Environment","date":"2019-05-23","arxiv_id":"1905.09673","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-semantics-to-execution-integrating","title":"From semantics to execution: Integrating action planning with reinforcement learning for robotic causal problem-solving","date":"2019-05-23","arxiv_id":"1905.09683","repositories_listed":0,"syntology":null},{"url":null,"slug":"pac-guarantees-for-concurrent-reinforcement","title":"PAC Guarantees for Cooperative Multi-Agent Reinforcement Learning with Restricted Communication","date":"2019-05-23","arxiv_id":"1905.09951","repositories_listed":0,"syntology":null},{"url":null,"slug":"population-based-global-optimisation-methods","title":"Population-based Global Optimisation Methods for Learning Long-term Dependencies with RNNs","date":"2019-05-23","arxiv_id":"1905.09691","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-detecting","title":"Deep Reinforcement Learning for Detecting Malicious Websites","date":"2019-05-22","arxiv_id":"1905.09207","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-reinforcement-learning-for","title":"Hierarchical Reinforcement Learning for Quadruped Locomotion","date":"2019-05-22","arxiv_id":"1905.08926","repositories_listed":0,"syntology":null},{"url":null,"slug":"stochastic-inverse-reinforcement-learning","title":"Stochastic Inverse Reinforcement Learning","date":"2019-05-21","arxiv_id":"1905.08513","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-bayesian-approach-to-robust-reinforcement","title":"A Bayesian Approach to Robust Reinforcement Learning","date":"2019-05-20","arxiv_id":"1905.08188","repositories_listed":0,"syntology":null},{"url":null,"slug":"issues-concerning-realizability-of-blackwell","title":"Issues concerning realizability of Blackwell optimal policies in reinforcement learning","date":"2019-05-20","arxiv_id":"1905.08293","repositories_listed":0,"syntology":null},{"url":null,"slug":"perceptual-values-from-observation","title":"Perceptual Values from Observation","date":"2019-05-20","arxiv_id":"1905.07861","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-without-ground-truth","title":"Reinforcement Learning without Ground-Truth State","date":"2019-05-20","arxiv_id":"1905.07866","repositories_listed":0,"syntology":null},{"url":null,"slug":"stochastic-variance-reduction-for-deep-q","title":"Stochastic Variance Reduction for Deep Q-learning","date":"2019-05-20","arxiv_id":"1905.08152","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-learning-of","title":"Reinforcement Learning for Learning of Dynamical Systems in Uncertain Environment: a Tutorial","date":"2019-05-19","arxiv_id":"1905.07727","repositories_listed":0,"syntology":null},{"url":null,"slug":"evolving-rewards-to-automate-reinforcement","title":"Evolving Rewards to Automate Reinforcement Learning","date":"2019-05-18","arxiv_id":"1905.07628","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-channel","title":"Deep Reinforcement Learning-Based Channel Allocation for Wireless LANs with Graph Convolutional Networks","date":"2019-05-17","arxiv_id":"1905.07144","repositories_listed":0,"syntology":null},{"url":null,"slug":"enforcing-constraints-for-time-series","title":"Enforcing constraints for time series prediction in supervised, unsupervised and reinforcement learning","date":"2019-05-17","arxiv_id":"1905.07501","repositories_listed":0,"syntology":null},{"url":null,"slug":"in-support-of-over-parametrization-in-deep","title":"In Support of Over-Parametrization in Deep Reinforcement Learning: an Empirical Study","date":"2019-05-17","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"mamic-macro-and-micro-curriculum-for-robotic","title":"MaMiC: Macro and Micro Curriculum for Robotic Reinforcement Learning","date":"2019-05-17","arxiv_id":"1905.07193","repositories_listed":0,"syntology":null},{"url":null,"slug":"stochastically-dominant-distributional","title":"Stochastically Dominant Distributional Reinforcement Learning","date":"2019-05-17","arxiv_id":"1905.07318","repositories_listed":0,"syntology":null},{"url":null,"slug":"stratospheric-aerosol-injection-as-a-deep","title":"Stratospheric Aerosol Injection as a Deep Reinforcement Learning Problem","date":"2019-05-17","arxiv_id":"1905.07366","repositories_listed":0,"syntology":null},{"url":null,"slug":"tbq-improving-efficiency-of-trace-utilization","title":"TBQ($σ$): Improving Efficiency of Trace Utilization for Off-Policy Reinforcement Learning","date":"2019-05-17","arxiv_id":"1905.07237","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-knowledge-based-agent-learning-to-do","title":"Deep Knowledge Based Agent: Learning to do tasks by self-thinking about imaginary worlds","date":"2019-05-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-based-sequential-decision-making","title":"Knowledge-Based Sequential Decision-Making Under Uncertainty","date":"2019-05-16","arxiv_id":"1905.07030","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-exploration-policies-for-model","title":"Learning Exploration Policies for Model-Agnostic Meta-Reinforcement Learning","date":"2019-05-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-for-adaptive-1","title":"Meta-Reinforcement Learning for Adaptive Autonomous Driving","date":"2019-05-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"sub-policy-adaptation-for-hierarchical-1","title":"Sub-policy Adaptation for Hierarchical Reinforcement Learning","date":"2019-05-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-learning-based-branch-and-bound-for-maximum","title":"A Learning based Branch and Bound for Maximum Common Subgraph Problems","date":"2019-05-15","arxiv_id":"1905.05840","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-penetration-testing-using","title":"Autonomous Penetration Testing using Reinforcement Learning","date":"2019-05-15","arxiv_id":"1905.05965","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-scheduling-in","title":"Deep reinforcement learning for scheduling in large-scale networked control systems","date":"2019-05-15","arxiv_id":"1905.05992","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-scheduling-in-1","title":"Deep Reinforcement Learning for Scheduling in Cellular Networks","date":"2019-05-15","arxiv_id":"1905.05914","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-robotics-and","title":"Reinforcement Learning for Robotics and Control with Active Uncertainty Reduction","date":"2019-05-15","arxiv_id":"1905.06274","repositories_listed":0,"syntology":null},{"url":null,"slug":"combining-parametric-and-nonparametric-models","title":"Combining Parametric and Nonparametric Models for Off-Policy Evaluation","date":"2019-05-14","arxiv_id":"1905.05787","repositories_listed":0,"syntology":null},{"url":null,"slug":"tauriel-targeting-traveling-salesman-problem","title":"TauRieL: Targeting Traveling Salesman Problem with a deep reinforcement learning inspired architecture","date":"2019-05-14","arxiv_id":"1905.05567","repositories_listed":0,"syntology":null},{"url":null,"slug":"variational-regret-bounds-for-reinforcement","title":"Variational Regret Bounds for Reinforcement Learning","date":"2019-05-14","arxiv_id":"1905.05857","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-multi-agent-reinforcement-learning-based","title":"Deep Multi-Agent Reinforcement Learning Based Cooperative Edge Caching in Wireless Networks","date":"2019-05-13","arxiv_id":"1905.05256","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributional-reinforcement-learning-for","title":"Distributional Reinforcement Learning for Efficient Exploration","date":"2019-05-13","arxiv_id":"1905.06125","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-and-exploiting-multiple-subgoals-for","title":"Learning and Exploiting Multiple Subgoals for Fast Exploration in Hierarchical Reinforcement Learning","date":"2019-05-13","arxiv_id":"1905.05180","repositories_listed":0,"syntology":null},{"url":null,"slug":"diagnosing-reinforcement-learning-for-traffic","title":"Diagnosing Reinforcement Learning for Traffic Signal Control","date":"2019-05-12","arxiv_id":"1905.04716","repositories_listed":0,"syntology":null},{"url":null,"slug":"metareasoning-in-modular-software-systems-on","title":"Metareasoning in Modular Software Systems: On-the-Fly Configuration using Reinforcement Learning with Rich Contextual Representations","date":"2019-05-12","arxiv_id":"1905.05179","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-routerless-network-on-chip-designs","title":"Optimizing Routerless Network-on-Chip Designs: An Innovative Learning-Based Framework","date":"2019-05-11","arxiv_id":"1905.04423","repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-based-deep-reinforcement-learning","title":"Attention-based Deep Reinforcement Learning for Multi-view Environments","date":"2019-05-10","arxiv_id":"1905.03985","repositories_listed":0,"syntology":null},{"url":null,"slug":"design-of-artificial-intelligence-agents-for","title":"Design of Artificial Intelligence Agents for Games using Deep Reinforcement Learning","date":"2019-05-10","arxiv_id":"1905.04127","repositories_listed":0,"syntology":null},{"url":null,"slug":"do-autonomous-agents-benefit-from-hearing","title":"Do Autonomous Agents Benefit from Hearing?","date":"2019-05-10","arxiv_id":"1905.04192","repositories_listed":0,"syntology":null},{"url":null,"slug":"domain-adversarial-reinforcement-learning-for","title":"Domain Adversarial Reinforcement Learning for Partial Domain Adaptation","date":"2019-05-10","arxiv_id":"1905.04094","repositories_listed":0,"syntology":null},{"url":null,"slug":"emergent-escape-based-flocking-behavior-using","title":"Emergent Escape-based Flocking Behavior using Multi-Agent Reinforcement Learning","date":"2019-05-10","arxiv_id":"1905.04077","repositories_listed":0,"syntology":null},{"url":null,"slug":"gan-based-deep-distributional-reinforcement","title":"GAN-powered Deep Distributional Reinforcement Learning for Resource Management in Network Slicing","date":"2019-05-10","arxiv_id":"1905.03929","repositories_listed":0,"syntology":null},{"url":null,"slug":"intelligent-user-association-for-symbiotic","title":"Intelligent User Association for Symbiotic Radio Networks using Deep Reinforcement Learning","date":"2019-05-10","arxiv_id":"1905.04041","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-detection-of-mutual-influences-and","title":"On the Detection of Mutual Influences and Their Consideration in Reinforcement Learning Processes","date":"2019-05-10","arxiv_id":"1905.04205","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-non-stationary","title":"Reinforcement Learning in Non-Stationary Environments","date":"2019-05-10","arxiv_id":"1905.03970","repositories_listed":0,"syntology":null},{"url":null,"slug":"190503440","title":"Path Design for Cellular-Connected UAV with Reinforcement Learning","date":"2019-05-09","arxiv_id":"1905.03440","repositories_listed":0,"syntology":null},{"url":null,"slug":"190503494","title":"Toward Packet Routing with Fully-distributed Multi-agent Deep Reinforcement Learning","date":"2019-05-09","arxiv_id":"1905.03494","repositories_listed":0,"syntology":null},{"url":null,"slug":"190503501","title":"Pretrain Soft Q-Learning with Imperfect Demonstrations","date":"2019-05-09","arxiv_id":"1905.03501","repositories_listed":0,"syntology":null},{"url":null,"slug":"190503726","title":"A Reinforcement Learning Perspective on the Optimal Control of Mutation Probabilities for the (1+1) Evolutionary Algorithm: First Results on the OneMax Problem","date":"2019-05-09","arxiv_id":"1905.03726","repositories_listed":0,"syntology":null},{"url":null,"slug":"190508314","title":"Longitudinal Dynamic versus Kinematic Models for Car-Following Control Using Deep Reinforcement Learning","date":"2019-05-07","arxiv_id":"1905.08314","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerated-target-updates-for-q-learning","title":"Accelerated Target Updates for Q-learning","date":"2019-05-07","arxiv_id":"1905.02841","repositories_listed":0,"syntology":null},{"url":null,"slug":"continual-and-multi-task-reinforcement","title":"Continual and Multi-task Reinforcement Learning With Shared Episodic Memory","date":"2019-05-07","arxiv_id":"1905.02662","repositories_listed":0,"syntology":null},{"url":null,"slug":"object-exchangeability-in-reinforcement","title":"Object Exchangeability in Reinforcement Learning: Extended Abstract","date":"2019-05-07","arxiv_id":"1905.02698","repositories_listed":0,"syntology":null},{"url":null,"slug":"regal-transfer-learning-for-fast-optimization","title":"Reinforced Genetic Algorithm Learning for Optimizing Computation Graphs","date":"2019-05-07","arxiv_id":"1905.02494","repositories_listed":0,"syntology":null},{"url":null,"slug":"190511437","title":"A Survey of Adaptive Resonance Theory Neural Network Models for Engineering Applications","date":"2019-05-04","arxiv_id":"1905.11437","repositories_listed":0,"syntology":null},{"url":null,"slug":"face-hallucination-by-attentive-sequence","title":"Face Hallucination by Attentive Sequence Optimization with Reinforcement Learning","date":"2019-05-04","arxiv_id":"1905.01509","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-intelligent-secondary-control-of","title":"Adaptive Intelligent Secondary Control of Microgrids Using a Biologically-Inspired Reinforcement Learning","date":"2019-05-02","arxiv_id":"1905.00557","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-air-traffic-controller-a-deep","title":"Autonomous Air Traffic Controller: A Deep Multi-Agent Reinforcement Learning Approach","date":"2019-05-02","arxiv_id":"1905.01303","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-guider-network-for-multi-dual-learning","title":"A Guider Network for Multi-Dual Learning","date":"2019-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-new-dog-learns-old-tricks-rl-finds-classic","title":"A new dog learns old tricks: RL finds classic optimization algorithms","date":"2019-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"actrce-augmenting-experience-via-teachers-1","title":"ACTRCE: Augmenting Experience via Teacher’s Advice","date":"2019-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"automata-guided-skill-composition","title":"Automata Guided Skill Composition","date":"2019-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-relational","title":"Deep reinforcement learning with relational inductive biases","date":"2019-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"driving-with-style-inverse-reinforcement","title":"Driving with Style: Inverse Reinforcement Learning in General-Purpose Planning for Automated Driving","date":"2019-05-01","arxiv_id":"1905.00229","repositories_listed":0,"syntology":null},{"url":null,"slug":"explicit-recall-for-efficient-exploration","title":"Explicit Recall for Efficient Exploration","date":"2019-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"few-shot-intent-inference-via-meta-inverse","title":"Few-Shot Intent Inference via Meta-Inverse Reinforcement Learning","date":"2019-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"information-theoretic-considerations-in-batch","title":"Information-Theoretic Considerations in Batch Reinforcement Learning","date":"2019-05-01","arxiv_id":"1905.00360","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-actionable-representations-with-goal-1","title":"Learning Actionable Representations with Goal Conditioned Policies","date":"2019-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-agents-with-prioritization-and","title":"Learning agents with prioritization and parameter noise in continuous state and action space","date":"2019-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-goal-conditioned-value-functions","title":"Learning Goal-Conditioned Value Functions with one-step Path rewards rather than Goal-Rewards","date":"2019-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-heuristics-for-automated-reasoning-1","title":"Learning Heuristics for Automated Reasoning through Reinforcement Learning","date":"2019-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null}],"record_sha256":"71a998bfd615b69e67acfdd27722eb87585d0fd9f06f81a1b439bfdb790ae77d","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}