{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/ppo/papers/9","list_of":"/method/ppo","method":"PPO","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":9,"pages_in_order":10,"rows_per_page":100,"rows":[801,900],"of":949,"counts":{"archive_papers_tagged":949,"with_a_code_link":397,"where_syntology_ran_a_sample":139,"not_listed_spam_title":0,"listed":949,"listed_where_code_ran":139,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":114,"every_run_a_failure_of_syntologys_instrument":25,"listed_with_a_run_with_no_instrument_failure":114,"listed_every_run_a_failure_of_syntologys_instrument":25,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/ppo","prev":"/method/ppo/papers/8","next":"/method/ppo/papers/10","papers":[{"paper":null,"slug":"proximal-policy-optimization-via-enhanced","title":"Proximal Policy Optimization via Enhanced Exploration Efficiency","date":"2020-11-11","arxiv_id":"2011.05525","n_code_links":0,"syntology":null},{"paper":"/paper/trajectory-planning-for-autonomous-vehicles","slug":"trajectory-planning-for-autonomous-vehicles","title":"Trajectory Planning for Autonomous Vehicles Using Hierarchical Reinforcement Learning","date":"2020-11-09","arxiv_id":"2011.04752","n_code_links":1,"syntology":null},{"paper":"/paper/multimodal-trajectory-prediction-via","slug":"multimodal-trajectory-prediction-via","title":"Multimodal Trajectory Prediction via Topological Invariance for Navigation at Uncontrolled Intersections","date":"2020-11-08","arxiv_id":"2011.03894","n_code_links":1,"syntology":null},{"paper":"/paper/drafting-in-collectible-card-games-via","slug":"drafting-in-collectible-card-games-via","title":"Drafting in Collectible Card Games via Reinforcement Learning","date":"2020-11-07","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/guided-dialogue-policy-learning-without","slug":"guided-dialogue-policy-learning-without","title":"Guided Dialogue Policy Learning without Adversarial Learning in the Loop","date":"2020-11-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"pilot-efficient-planning-by-imitation","title":"PILOT: Efficient Planning by Imitation Learning and Optimisation for Safe Autonomous Driving","date":"2020-11-01","arxiv_id":"2011.00509","n_code_links":0,"syntology":null},{"paper":"/paper/a-software-architecture-for-autonomous","slug":"a-software-architecture-for-autonomous","title":"A Software Architecture for Autonomous Vehicles: Team LRM-B Entry in the First CARLA Autonomous Driving Challenge","date":"2020-10-23","arxiv_id":"2010.12598","n_code_links":0,"syntology":null},{"paper":null,"slug":"proximal-policy-gradient-ppo-with-policy","title":"Proximal Policy Gradient: PPO with Policy Gradient","date":"2020-10-20","arxiv_id":"2010.09933","n_code_links":0,"syntology":null},{"paper":"/paper/finding-physical-adversarial-examples-for-1","slug":"finding-physical-adversarial-examples-for-1","title":"Finding Physical Adversarial Examples for Autonomous Driving with Fast and Differentiable Image Compositing","date":"2020-10-17","arxiv_id":"2010.08844","n_code_links":1,"syntology":null},{"paper":"/paper/learning-monocular-dense-depth-from-events","slug":"learning-monocular-dense-depth-from-events","title":"Learning Monocular Dense Depth from Events","date":"2020-10-16","arxiv_id":"2010.08350","n_code_links":1,"syntology":null},{"paper":null,"slug":"recurrent-distributed-reinforcement-learning","title":"A Learning Approach to Robot-Agnostic Force-Guided High Precision Assembly","date":"2020-10-15","arxiv_id":"2010.08052","n_code_links":0,"syntology":null},{"paper":"/paper/unsupervised-learning-of-depth-and-ego-motion-3","slug":"unsupervised-learning-of-depth-and-ego-motion-3","title":"Unsupervised Learning of Depth and Ego-Motion from Cylindrical Panoramic Video with Applications for Virtual Reality","date":"2020-10-14","arxiv_id":"2010.07704","n_code_links":1,"syntology":null},{"paper":null,"slug":"lm-reloc-levenberg-marquardt-based-direct","title":"LM-Reloc: Levenberg-Marquardt Based Direct Visual Relocalization","date":"2020-10-13","arxiv_id":"2010.06323","n_code_links":0,"syntology":null},{"paper":"/paper/discrete-latent-space-world-models-for","slug":"discrete-latent-space-world-models-for","title":"Smaller World Models for Reinforcement Learning","date":"2020-10-12","arxiv_id":"2010.05767","n_code_links":0,"syntology":null},{"paper":"/paper/automated-concatenation-of-embeddings-for-1","slug":"automated-concatenation-of-embeddings-for-1","title":"Automated Concatenation of Embeddings for Structured Prediction","date":"2020-10-10","arxiv_id":"2010.05006","n_code_links":2,"syntology":null},{"paper":null,"slug":"proximal-policy-optimization-with-relative","title":"Proximal Policy Optimization with Relative Pearson Divergence","date":"2020-10-07","arxiv_id":"2010.03290","n_code_links":0,"syntology":null},{"paper":"/paper/revisiting-design-choices-in-proximal-policy","slug":"revisiting-design-choices-in-proximal-policy","title":"Revisiting Design Choices in Proximal Policy Optimization","date":"2020-09-23","arxiv_id":"2009.10897","n_code_links":1,"syntology":{"ran":5,"of":11,"n_ran_checked":4,"n_instrument":1,"unverified":6,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","official":{"repos":["chloechsu/revisiting-ppo"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/phasic-policy-gradient","slug":"phasic-policy-gradient","title":"Phasic Policy Gradient","date":"2020-09-09","arxiv_id":"2009.04416","n_code_links":3,"syntology":{"ran":7,"of":13,"n_ran_checked":7,"n_instrument":0,"unverified":6,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","official":{"repos":["openai/phasic-policy-gradient"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"data-driven-transferred-energy-management","title":"Data-Driven Transferred Energy Management Strategy for Hybrid Electric Vehicles via Deep Reinforcement Learning","date":"2020-09-07","arxiv_id":"2009.03289","n_code_links":0,"syntology":null},{"paper":"/paper/drle-decentralized-reinforcement-learning-at","slug":"drle-decentralized-reinforcement-learning-at","title":"DRLE: Decentralized Reinforcement Learning at the Edge for Traffic Light Control in the IoV","date":"2020-09-03","arxiv_id":"2009.01502","n_code_links":1,"syntology":null},{"paper":null,"slug":"driving-through-ghosts-behavioral-cloning","title":"Driving Through Ghosts: Behavioral Cloning with False Positives","date":"2020-08-29","arxiv_id":"2008.12969","n_code_links":0,"syntology":null},{"paper":"/paper/domain-adaptation-through-task-distillation","slug":"domain-adaptation-through-task-distillation","title":"Domain Adaptation Through Task Distillation","date":"2020-08-27","arxiv_id":"2008.11911","n_code_links":1,"syntology":null},{"paper":"/paper/query-focused-multi-document-summarisation-of-1","slug":"query-focused-multi-document-summarisation-of-1","title":"Query Focused Multi-document Summarisation of Biomedical Texts: Macquarie Universiy and the Australian National University at BioASQ8b","date":"2020-08-27","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/reinforcement-learning-for-low-thrust","slug":"reinforcement-learning-for-low-thrust","title":"Reinforcement Learning for Low-Thrust Trajectory Design of Interplanetary Missions","date":"2020-08-19","arxiv_id":"2008.08501","n_code_links":1,"syntology":null},{"paper":"/paper/towards-closing-the-sim-to-real-gap-in","slug":"towards-closing-the-sim-to-real-gap-in","title":"Towards Closing the Sim-to-Real Gap in Collaborative Multi-Robot Deep Reinforcement Learning","date":"2020-08-18","arxiv_id":"2008.07875","n_code_links":1,"syntology":null},{"paper":null,"slug":"reinforced-wasserstein-training-for-severity","title":"Reinforced Wasserstein Training for Severity-Aware Semantic Segmentation in Autonomous Driving","date":"2020-08-11","arxiv_id":"2008.04751","n_code_links":0,"syntology":null},{"paper":null,"slug":"physical-adversarial-attack-on-vehicle","title":"Physical Adversarial Attack on Vehicle Detector in the Carla Simulator","date":"2020-07-31","arxiv_id":"2007.16118","n_code_links":0,"syntology":null},{"paper":"/paper/queueing-network-controls-via-deep","slug":"queueing-network-controls-via-deep","title":"Queueing Network Controls via Deep Reinforcement Learning","date":"2020-07-31","arxiv_id":"2008.01644","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/nsganetv2-evolutionary-multi-objective","slug":"nsganetv2-evolutionary-multi-objective","title":"NSGANetV2: Evolutionary Multi-Objective Surrogate-Assisted Neural Architecture Search","date":"2020-07-20","arxiv_id":"2007.10396","n_code_links":1,"syntology":{"ran":4,"of":8,"n_ran_checked":3,"n_instrument":1,"unverified":4,"pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","official":{"repos":["mikelzc1990/nsganetv2"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/developmental-reinforcement-learning-of","slug":"developmental-reinforcement-learning-of","title":"Developmental Reinforcement Learning of Control Policy of a Quadcopter UAV with Thrust Vectoring Rotors","date":"2020-07-15","arxiv_id":"2007.07793","n_code_links":1,"syntology":null},{"paper":null,"slug":"directional-primitives-for-uncertainty-aware","title":"Directional Primitives for Uncertainty-Aware Motion Estimation in Urban Environments","date":"2020-07-01","arxiv_id":"2007.00161","n_code_links":0,"syntology":null},{"paper":"/paper/sample-factory-egocentric-3d-control-from","slug":"sample-factory-egocentric-3d-control-from","title":"Sample Factory: Egocentric 3D Control from Pixels at 100000 FPS with Asynchronous Reinforcement Learning","date":"2020-06-21","arxiv_id":"2006.11751","n_code_links":4,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["alex-petrenko/sample-factory"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"an-operator-view-of-policy-gradient-methods","title":"An operator view of policy gradient methods","date":"2020-06-19","arxiv_id":"2006.11266","n_code_links":0,"syntology":null},{"paper":null,"slug":"generalization-of-agent-behavior-through","title":"Generalization of Agent Behavior through Explicit Representation of Context","date":"2020-06-18","arxiv_id":"2006.11305","n_code_links":0,"syntology":null},{"paper":"/paper/fine-tuning-darts-for-image-classification","slug":"fine-tuning-darts-for-image-classification","title":"Fine-Tuning DARTS for Image Classification","date":"2020-06-16","arxiv_id":"2006.09042","n_code_links":0,"syntology":null},{"paper":null,"slug":"shieldnn-a-provably-safe-nn-filter-for-unsafe","title":"ShieldNN: A Provably Safe NN Filter for Unsafe NN Controllers","date":"2020-06-16","arxiv_id":"2006.09564","n_code_links":0,"syntology":null},{"paper":"/paper/optimistic-distributionally-robust-policy","slug":"optimistic-distributionally-robust-policy","title":"Optimistic Distributionally Robust Policy Optimization","date":"2020-06-14","arxiv_id":"2006.07815","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["kadysongbb/dr-trpo"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/bonsai-net-one-shot-neural-architecture","slug":"bonsai-net-one-shot-neural-architecture","title":"Bonsai-Net: One-Shot Neural Architecture Search via Differentiable Pruners","date":"2020-06-12","arxiv_id":"2006.09264","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["RobGeada/bonsai-net-lite"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"exploration-by-maximizing-renyi-entropy-for","title":"Exploration by Maximizing Rényi Entropy for Reward-Free RL Framework","date":"2020-06-11","arxiv_id":"2006.06193","n_code_links":0,"syntology":null},{"paper":"/paper/rethinking-pre-training-and-self-training","slug":"rethinking-pre-training-and-self-training","title":"Rethinking Pre-training and Self-training","date":"2020-06-11","arxiv_id":"2006.06882","n_code_links":2,"syntology":null},{"paper":null,"slug":"learning-navigation-costs-from-demonstration-1","title":"Learning Navigation Costs from Demonstration with Semantic Observations","date":"2020-06-09","arxiv_id":"2006.05043","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-comparison-of-self-play-algorithms-under-a","title":"A Comparison of Self-Play Algorithms Under a Generalized Framework","date":"2020-06-08","arxiv_id":"2006.04471","n_code_links":0,"syntology":null},{"paper":null,"slug":"fast-synthetic-lidar-rendering-via-spherical","title":"Fast Synthetic LiDAR Rendering via Spherical UV Unwrapping of Equirectangular Z-Buffer Images","date":"2020-06-08","arxiv_id":"2006.04345","n_code_links":0,"syntology":null},{"paper":null,"slug":"explaining-autonomous-driving-by-learning-end","title":"Explaining Autonomous Driving by Learning End-to-End Visual Attention","date":"2020-06-05","arxiv_id":"2006.03347","n_code_links":0,"syntology":null},{"paper":"/paper/optimization-and-passive-flow-control-using","slug":"optimization-and-passive-flow-control-using","title":"Single-step deep reinforcement learning for open-loop control of laminar and turbulent flows","date":"2020-06-04","arxiv_id":"2006.02979","n_code_links":1,"syntology":null},{"paper":"/paper/exploring-data-aggregation-in-policy-learning","slug":"exploring-data-aggregation-in-policy-learning","title":"Exploring Data Aggregation in Policy Learning for Vision-Based Urban Autonomous Driving","date":"2020-06-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-situational-driving","title":"Learning Situational Driving","date":"2020-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"severity-aware-semantic-segmentation-with","title":"Severity-Aware Semantic Segmentation With Reinforced Wasserstein Training","date":"2020-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/fast-risk-assessment-for-autonomous-vehicles","slug":"fast-risk-assessment-for-autonomous-vehicles","title":"Fast Risk Assessment for Autonomous Vehicles Using Learned Models of Agent Futures","date":"2020-05-27","arxiv_id":"2005.13458","n_code_links":1,"syntology":null},{"paper":null,"slug":"dynamic-value-estimation-for-single-task","title":"Dynamic Value Estimation for Single-Task Multi-Scene Reinforcement Learning","date":"2020-05-25","arxiv_id":"2005.12254","n_code_links":0,"syntology":null},{"paper":"/paper/implementation-matters-in-deep-policy","slug":"implementation-matters-in-deep-policy","title":"Implementation Matters in Deep Policy Gradients: A Case Study on PPO and TRPO","date":"2020-05-25","arxiv_id":"2005.12729","n_code_links":3,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["MadryLab/implementation-matters"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/mirror-descent-policy-optimization","slug":"mirror-descent-policy-optimization","title":"Mirror Descent Policy Optimization","date":"2020-05-20","arxiv_id":"2005.09814","n_code_links":1,"syntology":null},{"paper":"/paper/generalized-state-dependent-exploration-for","slug":"generalized-state-dependent-exploration-for","title":"Smooth Exploration for Robotic Reinforcement Learning","date":"2020-05-12","arxiv_id":"2005.05719","n_code_links":4,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["DLR-RM/stable-baselines3"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/learning-hierarchical-behavior-and-motion","slug":"learning-hierarchical-behavior-and-motion","title":"Learning hierarchical behavior and motion planning for autonomous driving","date":"2020-05-08","arxiv_id":"2005.03863","n_code_links":1,"syntology":null},{"paper":null,"slug":"model-based-reinforcement-learning-for","title":"Model-based reinforcement learning for biological sequence design","date":"2020-05-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/reinforcement-learning-with-augmented-data","slug":"reinforcement-learning-with-augmented-data","title":"Reinforcement Learning with Augmented Data","date":"2020-04-30","arxiv_id":"2004.14990","n_code_links":2,"syntology":{"ran":21,"of":24,"n_ran_checked":9,"n_instrument":12,"unverified":3,"pointer_only":21,"phrase":"21 ran (of which 8 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 12 where Syntology's instrument failed) · 3 unverified","official":null}},{"paper":"/paper/per-step-reward-a-new-perspective-for-risk","slug":"per-step-reward-a-new-perspective-for-risk","title":"Mean-Variance Policy Iteration for Risk-Averse Reinforcement Learning","date":"2020-04-22","arxiv_id":"2004.10888","n_code_links":1,"syntology":null},{"paper":null,"slug":"parkpredict-motion-and-intent-prediction-of","title":"ParkPredict: Motion and Intent Prediction of Vehicles in Parking Lots","date":"2020-04-21","arxiv_id":"2004.10293","n_code_links":0,"syntology":null},{"paper":"/paper/guided-dialog-policy-learning-without","slug":"guided-dialog-policy-learning-without","title":"Guided Dialog Policy Learning without Adversarial Learning in the Loop","date":"2020-04-07","arxiv_id":"2004.03267","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["cszmli/dp-without-adv"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"paper":"/paper/obstacle-avoidance-and-navigation-utilizing","slug":"obstacle-avoidance-and-navigation-utilizing","title":"Obstacle Avoidance and Navigation Utilizing Reinforcement Learning with Reward Shaping","date":"2020-03-28","arxiv_id":"2003.12863","n_code_links":1,"syntology":null},{"paper":"/paper/towards-safer-self-driving-through-great-pain","slug":"towards-safer-self-driving-through-great-pain","title":"Towards Safer Self-Driving Through Great PAIN (Physically Adversarial Intelligent Networks)","date":"2020-03-24","arxiv_id":"2003.10662","n_code_links":1,"syntology":null},{"paper":"/paper/robust-deep-reinforcement-learning-against","slug":"robust-deep-reinforcement-learning-against","title":"Robust Deep Reinforcement Learning against Adversarial Perturbations on State Observations","date":"2020-03-19","arxiv_id":"2003.08938","n_code_links":4,"syntology":null},{"paper":"/paper/particle-based-adaptive-discretization-for","slug":"particle-based-adaptive-discretization-for","title":"PFPN: Continuous Control of Physically Simulated Characters using Particle Filtering Policy Network","date":"2020-03-16","arxiv_id":"2003.06959","n_code_links":1,"syntology":null},{"paper":"/paper/fast-online-adaptation-in-robotics-through","slug":"fast-online-adaptation-in-robotics-through","title":"Fast Online Adaptation in Robotics through Meta-Learning Embeddings of Simulated Priors","date":"2020-03-10","arxiv_id":"2003.04663","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-machine-learning-environment-for-evaluating","title":"A machine learning environment for evaluating autonomous driving software","date":"2020-03-07","arxiv_id":"2003.03576","n_code_links":0,"syntology":null},{"paper":"/paper/reinforcement-learning-framework-for-deep","slug":"reinforcement-learning-framework-for-deep","title":"Reinforcement Learning Framework for Deep Brain Stimulation Study","date":"2020-02-22","arxiv_id":"2002.10948","n_code_links":1,"syntology":null},{"paper":"/paper/first-order-optimization-in-policy-space-for","slug":"first-order-optimization-in-policy-space-for","title":"First Order Constrained Optimization in Policy Space","date":"2020-02-16","arxiv_id":"2002.06506","n_code_links":2,"syntology":null},{"paper":"/paper/deep-rl-agent-for-a-real-time-action-strategy","slug":"deep-rl-agent-for-a-real-time-action-strategy","title":"Deep RL Agent for a Real-Time Action Strategy Game","date":"2020-02-15","arxiv_id":"2002.06290","n_code_links":1,"syntology":null},{"paper":null,"slug":"temporal-adaptive-hierarchical-reinforcement","title":"Temporal-adaptive Hierarchical Reinforcement Learning","date":"2020-02-06","arxiv_id":"2002.02080","n_code_links":0,"syntology":null},{"paper":"/paper/integrating-deep-reinforcement-learning-with","slug":"integrating-deep-reinforcement-learning-with","title":"Integrating Deep Reinforcement Learning with Model-based Path Planners for Automated Driving","date":"2020-02-02","arxiv_id":"2002.00434","n_code_links":1,"syntology":{"ran":2,"of":5,"n_ran_checked":2,"n_instrument":0,"unverified":3,"pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["Ekim-Yurtsever/Hybrid-DeepRL-Automated-Driving"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/interpretable-end-to-end-urban-autonomous","slug":"interpretable-end-to-end-urban-autonomous","title":"Interpretable End-to-end Urban Autonomous Driving with Latent Deep Reinforcement Learning","date":"2020-01-23","arxiv_id":"2001.08726","n_code_links":4,"syntology":{"ran":2,"of":6,"n_ran_checked":2,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["cjy1992/interp-e2e-driving"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/continuous-action-reinforcement-learning-for","slug":"continuous-action-reinforcement-learning-for","title":"Continuous-action Reinforcement Learning for Playing Racing Games: Comparing SPG to PPO","date":"2020-01-15","arxiv_id":"2001.05270","n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-representations-in-reinforcement-1","title":"Learning Representations in Reinforcement Learning: an Information Bottleneck Approach","date":"2020-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"tpo-tree-search-policy-optimization-for","title":"TPO: TREE SEARCH POLICY OPTIMIZATION FOR CONTINUOUS ACTION SPACES","date":"2020-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/slm-lab-a-comprehensive-benchmark-and-modular-1","slug":"slm-lab-a-comprehensive-benchmark-and-modular-1","title":"SLM Lab: A Comprehensive Benchmark and Modular Software Framework for Reproducible Deep Reinforcement Learning","date":"2019-12-28","arxiv_id":"1912.12482","n_code_links":1,"syntology":{"ran":10,"of":13,"n_ran_checked":10,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["kengz/SLM-Lab"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/learning-by-cheating","slug":"learning-by-cheating","title":"Learning by Cheating","date":"2019-12-27","arxiv_id":"1912.12294","n_code_links":9,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["dotchen/LearningByCheating"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"mastering-complex-control-in-moba-games-with","title":"Mastering Complex Control in MOBA Games with Deep Reinforcement Learning","date":"2019-12-20","arxiv_id":"1912.09729","n_code_links":0,"syntology":null},{"paper":"/paper/spinenet-learning-scale-permuted-backbone-for","slug":"spinenet-learning-scale-permuted-backbone-for","title":"SpineNet: Learning Scale-Permuted Backbone for Recognition and Localization","date":"2019-12-10","arxiv_id":"1912.05027","n_code_links":13,"syntology":null},{"paper":"/paper/lates-latent-space-distillation-for-teacher","slug":"lates-latent-space-distillation-for-teacher","title":"SAM: Squeeze-and-Mimic Networks for Conditional Visual Driving Policy Learning","date":"2019-12-06","arxiv_id":"1912.02973","n_code_links":1,"syntology":null},{"paper":"/paper/mnasfpn-learning-latency-aware-pyramid","slug":"mnasfpn-learning-latency-aware-pyramid","title":"MnasFPN: Learning Latency-aware Pyramid Architecture for Object Detection on Mobile Devices","date":"2019-12-02","arxiv_id":"1912.01106","n_code_links":2,"syntology":null},{"paper":null,"slug":"on-policy-reinforcement-learning-with-entropy","title":"Policy Optimization Reinforcement Learning with Entropy Regularization","date":"2019-12-02","arxiv_id":"1912.01557","n_code_links":0,"syntology":null},{"paper":"/paper/automated-curriculum-generation-for-policy","slug":"automated-curriculum-generation-for-policy","title":"Automated curriculum generation for Policy Gradients from Demonstrations","date":"2019-12-01","arxiv_id":"1912.00444","n_code_links":1,"syntology":null},{"paper":"/paper/learning-reward-machines-for-partially","slug":"learning-reward-machines-for-partially","title":"Learning Reward Machines for Partially Observable Reinforcement Learning","date":"2019-12-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"neural-trust-regionproximal-policy","title":"Neural Trust Region/Proximal Policy Optimization Attains Globally Optimal Policy","date":"2019-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"impact-importance-weighted-asynchronous-1","title":"IMPACT: Importance Weighted Asynchronous Architectures with Clipped Target Networks","date":"2019-11-30","arxiv_id":"1912.00167","n_code_links":0,"syntology":null},{"paper":"/paper/end-to-end-model-free-reinforcement-learning","slug":"end-to-end-model-free-reinforcement-learning","title":"End-to-End Model-Free Reinforcement Learning for Urban Driving using Implicit Affordances","date":"2019-11-25","arxiv_id":"1911.10868","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":4,"n_instrument":1,"unverified":2,"pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["valeoai/LearningByCheating"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"unsupervised-neural-sensor-models-for","title":"Unsupervised Neural Sensor Models for Synthetic LiDAR Data Augmentation","date":"2019-11-24","arxiv_id":"1911.10575","n_code_links":0,"syntology":null},{"paper":null,"slug":"accelerating-training-in-pommerman-with","title":"Accelerating Training in Pommerman with Imitation and Reinforcement Learning","date":"2019-11-12","arxiv_id":"1911.04947","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-representations-in-reinforcement","title":"Learning Representations in Reinforcement Learning:An Information Bottleneck Approach","date":"2019-11-12","arxiv_id":"1911.05695","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-modal-semantic-segmentation-using","title":"Multi Modal Semantic Segmentation using Synthetic Data","date":"2019-10-30","arxiv_id":"1910.13676","n_code_links":0,"syntology":null},{"paper":"/paper/hrl4in-hierarchical-reinforcement-learning","slug":"hrl4in-hierarchical-reinforcement-learning","title":"HRL4IN: Hierarchical Reinforcement Learning for Interactive Navigation with Mobile Manipulators","date":"2019-10-24","arxiv_id":"1910.11432","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"conditional-driving-from-natural-language","title":"Conditional Driving from Natural Language Instructions","date":"2019-10-16","arxiv_id":"1910.07615","n_code_links":0,"syntology":null},{"paper":"/paper/quantized-reinforcement-learning-quarl","slug":"quantized-reinforcement-learning-quarl","title":"QuaRL: Quantization for Fast and Environmentally Sustainable Reinforcement Learning","date":"2019-10-02","arxiv_id":"1910.01055","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["harvard-edge/quarl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"geometry-aware-video-object-detection-for","title":"Geometry-Aware Video Object Detection for Static Cameras","date":"2019-09-06","arxiv_id":"1909.03140","n_code_links":0,"syntology":null},{"paper":null,"slug":"conditional-vehicle-trajectories-prediction","title":"Conditional Vehicle Trajectories Prediction in CARLA Urban Environment","date":"2019-09-02","arxiv_id":"1909.00792","n_code_links":0,"syntology":null},{"paper":"/paper/doorgym-a-scalable-door-opening-environment","slug":"doorgym-a-scalable-door-opening-environment","title":"DoorGym: A Scalable Door Opening Environment And Baseline Agent","date":"2019-08-05","arxiv_id":"1908.01887","n_code_links":1,"syntology":null},{"paper":"/paper/towards-model-based-reinforcement-learning","slug":"towards-model-based-reinforcement-learning","title":"Towards Model-based Reinforcement Learning for Industry-near Environments","date":"2019-07-27","arxiv_id":"1907.11971","n_code_links":1,"syntology":null},{"paper":"/paper/google-research-football-a-novel","slug":"google-research-football-a-novel","title":"Google Research Football: A Novel Reinforcement Learning Environment","date":"2019-07-25","arxiv_id":"1907.11180","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["google-research/football"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/ppo-dash-improving-generalization-in-deep","slug":"ppo-dash-improving-generalization-in-deep","title":"PPO Dash: Improving Generalization in Deep Reinforcement Learning","date":"2019-07-15","arxiv_id":"1907.06704","n_code_links":1,"syntology":null},{"paper":null,"slug":"robust-guarantees-for-perception-based","title":"Robust Guarantees for Perception-Based Control","date":"2019-07-08","arxiv_id":"1907.03680","n_code_links":0,"syntology":null}],"record_sha256":"6370b98781674522235bb81c85f0e0807a730ad9761a075ae5f7f1515fec9273","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}