{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/ppo/papers/7","list_of":"/method/ppo","method":"PPO","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":7,"pages_in_order":10,"rows_per_page":100,"rows":[601,700],"of":949,"counts":{"archive_papers_tagged":949,"with_a_code_link":397,"where_syntology_ran_a_sample":139,"not_listed_spam_title":0,"listed":949,"listed_where_code_ran":139,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":114,"every_run_a_failure_of_syntologys_instrument":25,"listed_with_a_run_with_no_instrument_failure":114,"listed_every_run_a_failure_of_syntologys_instrument":25,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/ppo","prev":"/method/ppo/papers/6","next":"/method/ppo/papers/8","papers":[{"paper":"/paper/carla-gear-a-dataset-generator-for-a","slug":"carla-gear-a-dataset-generator-for-a","title":"CARLA-GeAR: a Dataset Generator for a Systematic Evaluation of Adversarial Robustness of Vision Models","date":"2022-06-09","arxiv_id":"2206.04365","n_code_links":1,"syntology":null},{"paper":"/paper/generalized-data-distribution-iteration","slug":"generalized-data-distribution-iteration","title":"Generalized Data Distribution Iteration","date":"2022-06-07","arxiv_id":"2206.03192","n_code_links":0,"syntology":null},{"paper":"/paper/gink-graph-based-interaction-aware","slug":"gink-graph-based-interaction-aware","title":"GIN: Graph-based Interaction-aware Constraint Policy Optimization for Autonomous Driving","date":"2022-06-03","arxiv_id":"2206.01488","n_code_links":1,"syntology":null},{"paper":null,"slug":"on-the-choice-of-data-for-efficient-training","title":"On the Choice of Data for Efficient Training and Validation of End-to-End Driving Models","date":"2022-06-01","arxiv_id":"2206.00608","n_code_links":0,"syntology":null},{"paper":"/paper/transfuser-imitation-with-transformer-based","slug":"transfuser-imitation-with-transformer-based","title":"TransFuser: Imitation with Transformer-Based Sensor Fusion for Autonomous Driving","date":"2022-05-31","arxiv_id":"2205.15997","n_code_links":3,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["autonomousvision/transfuser"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/efficient-reward-poisoning-attacks-on-online","slug":"efficient-reward-poisoning-attacks-on-online","title":"Efficient Reward Poisoning Attacks on Online Deep Reinforcement Learning","date":"2022-05-30","arxiv_id":"2205.14842","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yinglunxu/reward_poisoning_attack_drl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"risk-of-stochastic-systems-for-temporal-logic","title":"Risk of Stochastic Systems for Temporal Logic Specifications","date":"2022-05-28","arxiv_id":"2205.14523","n_code_links":0,"syntology":null},{"paper":"/paper/quark-controllable-text-generation-with","slug":"quark-controllable-text-generation-with","title":"Quark: Controllable Text Generation with Reinforced Unlearning","date":"2022-05-26","arxiv_id":"2205.13636","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["gximinglu/quark"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"learning-to-drive-using-sparse-imitation","title":"Learning to Drive Using Sparse Imitation Reinforcement Learning","date":"2022-05-24","arxiv_id":"2205.12128","n_code_links":0,"syntology":null},{"paper":"/paper/an-evaluation-study-of-intrinsic-motivation","slug":"an-evaluation-study-of-intrinsic-motivation","title":"An Evaluation Study of Intrinsic Motivation Techniques applied to Reinforcement Learning over Hard Exploration Environments","date":"2022-05-23","arxiv_id":"2205.11184","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":4,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["aklein1995/intrinsic_motivation_techniques_study"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/flexible-diffusion-modeling-of-long-videos","slug":"flexible-diffusion-modeling-of-long-videos","title":"Flexible Diffusion Modeling of Long Videos","date":"2022-05-23","arxiv_id":"2205.11495","n_code_links":1,"syntology":null},{"paper":null,"slug":"generalization-mayhems-and-limits-in","title":"Generalization, Mayhems and Limits in Recurrent Proximal Policy Optimization","date":"2022-05-23","arxiv_id":"2205.11104","n_code_links":0,"syntology":null},{"paper":"/paper/sigmoidally-preconditioned-off-policy","slug":"sigmoidally-preconditioned-off-policy","title":"The Sufficiency of Off-Policyness and Soft Clipping: PPO is still Insufficient according to an Off-Policy Measure","date":"2022-05-20","arxiv_id":"2205.10047","n_code_links":1,"syntology":null},{"paper":"/paper/a2c-is-a-special-case-of-ppo","slug":"a2c-is-a-special-case-of-ppo","title":"A2C is a special case of PPO","date":"2022-05-18","arxiv_id":"2205.09123","n_code_links":1,"syntology":null},{"paper":null,"slug":"constraining-the-attack-space-of-machine","title":"Defending Object Detectors against Patch Attacks with Out-of-Distribution Smoothing","date":"2022-05-18","arxiv_id":"2205.08989","n_code_links":0,"syntology":null},{"paper":null,"slug":"policy-distillation-with-selective-input","title":"Policy Distillation with Selective Input Gradient Regularization for Efficient Interpretability","date":"2022-05-18","arxiv_id":"2205.08685","n_code_links":0,"syntology":null},{"paper":null,"slug":"qualitative-differences-between-evolutionary","title":"Qualitative Differences Between Evolutionary Strategies and Reinforcement Learning Methods for Control of Autonomous Agents","date":"2022-05-16","arxiv_id":"2205.07592","n_code_links":0,"syntology":null},{"paper":"/paper/cliff-diving-exploring-reward-surfaces-in","slug":"cliff-diving-exploring-reward-surfaces-in","title":"Cliff Diving: Exploring Reward Surfaces in Reinforcement Learning Environments","date":"2022-05-14","arxiv_id":"2205.07015","n_code_links":0,"syntology":{"ran":2,"of":4,"n_ran_checked":1,"n_instrument":1,"unverified":2,"pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":null,"slug":"nmr-neural-manifold-representation-for","title":"NMR: Neural Manifold Representation for Autonomous Driving","date":"2022-05-11","arxiv_id":"2205.05551","n_code_links":0,"syntology":null},{"paper":null,"slug":"unrealnas-can-we-search-neural-architectures","title":"UnrealNAS: Can We Search Neural Architectures with Unreal Data?","date":"2022-05-04","arxiv_id":"2205.02162","n_code_links":0,"syntology":null},{"paper":null,"slug":"processing-network-controls-via-deep","title":"Processing Network Controls via Deep Reinforcement Learning","date":"2022-05-01","arxiv_id":"2205.02119","n_code_links":0,"syntology":null},{"paper":null,"slug":"control-aware-prediction-objectives-for","title":"Control-Aware Prediction Objectives for Autonomous Driving","date":"2022-04-28","arxiv_id":"2204.13319","n_code_links":0,"syntology":null},{"paper":"/paper/king-generating-safety-critical-driving","slug":"king-generating-safety-critical-driving","title":"KING: Generating Safety-Critical Driving Scenarios for Robust Imitation via Kinematics Gradients","date":"2022-04-28","arxiv_id":"2204.13683","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["autonomousvision/king"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"task-induced-representation-learning-1","title":"Task-Induced Representation Learning","date":"2022-04-25","arxiv_id":"2204.11827","n_code_links":0,"syntology":null},{"paper":null,"slug":"selfd-self-learning-large-scale-driving","title":"SelfD: Self-Learning Large-Scale Driving Policies From the Web","date":"2022-04-21","arxiv_id":"2204.10320","n_code_links":0,"syntology":null},{"paper":"/paper/deep-reinforcement-learning-for-a-two-echelon","slug":"deep-reinforcement-learning-for-a-two-echelon","title":"Comparing Deep Reinforcement Learning Algorithms in Two-Echelon Supply Chains","date":"2022-04-20","arxiv_id":"2204.09603","n_code_links":1,"syntology":null},{"paper":null,"slug":"selma-semantic-large-scale-multimodal","title":"SELMA: SEmantic Large-scale Multimodal Acquisitions in Variable Weather, Daytime and Viewpoints","date":"2022-04-20","arxiv_id":"2204.09788","n_code_links":0,"syntology":null},{"paper":"/paper/fully-end-to-end-autonomous-driving-with","slug":"fully-end-to-end-autonomous-driving-with","title":"End-to-end Autonomous Driving with Semantic Depth Cloud Mapping and Multi-agent","date":"2022-04-12","arxiv_id":"2204.05513","n_code_links":1,"syntology":null},{"paper":null,"slug":"proximal-policy-optimization-learning-based","title":"Proximal Policy Optimization Learning based Control of Congested Freeway Traffic","date":"2022-04-12","arxiv_id":"2204.05627","n_code_links":0,"syntology":null},{"paper":null,"slug":"scale-invariant-semantic-segmentation-with","title":"Scale Invariant Semantic Segmentation with RGB-D Fusion","date":"2022-04-10","arxiv_id":"2204.04679","n_code_links":0,"syntology":null},{"paper":null,"slug":"accelerating-federated-edge-learning-via-1","title":"Accelerating Federated Edge Learning via Topology Optimization","date":"2022-04-01","arxiv_id":"2204.00489","n_code_links":0,"syntology":null},{"paper":"/paper/hysteresis-based-rl-robustifying","slug":"hysteresis-based-rl-robustifying","title":"Hysteresis-Based RL: Robustifying Reinforcement Learning-based Control Policies via Hybrid Control","date":"2022-04-01","arxiv_id":"2204.00654","n_code_links":2,"syntology":null},{"paper":null,"slug":"assessing-evolutionary-terrain-generation","title":"Assessing Evolutionary Terrain Generation Methods for Curriculum Reinforcement Learning","date":"2022-03-29","arxiv_id":"2203.15172","n_code_links":0,"syntology":null},{"paper":"/paper/learning-from-all-vehicles","slug":"learning-from-all-vehicles","title":"Learning from All Vehicles","date":"2022-03-22","arxiv_id":"2203.11934","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":3,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["dotchen/LAV"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"proximal-policy-optimization-based-transmit","title":"Proximal Policy Optimization-based Transmit Beamforming and Phase-shift Design in an IRS-aided ISAC System for the THz Band","date":"2022-03-21","arxiv_id":"2203.10819","n_code_links":0,"syntology":null},{"paper":"/paper/microracer-a-didactic-environment-for-deep","slug":"microracer-a-didactic-environment-for-deep","title":"MicroRacer: a didactic environment for Deep Reinforcement Learning","date":"2022-03-20","arxiv_id":"2203.10494","n_code_links":1,"syntology":null},{"paper":"/paper/v2x-vit-vehicle-to-everything-cooperative","slug":"v2x-vit-vehicle-to-everything-cooperative","title":"V2X-ViT: Vehicle-to-Everything Cooperative Perception with Vision Transformer","date":"2022-03-20","arxiv_id":"2203.10638","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["DerrickXuNu/v2x-vit"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"dta-physical-camouflage-attacks-using","title":"DTA: Physical Camouflage Attacks using Differentiable Transformation Network","date":"2022-03-18","arxiv_id":"2203.09831","n_code_links":0,"syntology":null},{"paper":null,"slug":"proximal-policy-optimization-with-adaptive","title":"Proximal Policy Optimization with Adaptive Threshold for Symmetric Relative Density Ratio","date":"2022-03-18","arxiv_id":"2203.09809","n_code_links":0,"syntology":null},{"paper":null,"slug":"conquering-ghosts-relation-learning-for","title":"Conquering Ghosts: Relation Learning for Information Reliability Representation and End-to-End Robust Navigation","date":"2022-03-14","arxiv_id":"2203.09952","n_code_links":0,"syntology":null},{"paper":"/paper/synwoodscape-synthetic-surround-view-fisheye","slug":"synwoodscape-synthetic-surround-view-fisheye","title":"SynWoodScape: Synthetic Surround-view Fisheye Camera Dataset for Autonomous Driving","date":"2022-03-09","arxiv_id":"2203.05056","n_code_links":0,"syntology":null},{"paper":"/paper/mirror-differentiable-deep-social-projection","slug":"mirror-differentiable-deep-social-projection","title":"MIRROR: Differentiable Deep Social Projection for Assistive Human-Robot Communication","date":"2022-03-06","arxiv_id":"2203.02877","n_code_links":1,"syntology":null},{"paper":null,"slug":"ai-aided-traffic-control-scheme-for-m2m","title":"AI-aided Traffic Control Scheme for M2M Communications in the Internet of Vehicles","date":"2022-03-05","arxiv_id":"2204.03504","n_code_links":0,"syntology":null},{"paper":"/paper/risk-aware-scene-sampling-for-dynamic","slug":"risk-aware-scene-sampling-for-dynamic","title":"Risk-Aware Scene Sampling for Dynamic Assurance of Autonomous Systems","date":"2022-02-28","arxiv_id":"2202.13510","n_code_links":1,"syntology":null},{"paper":"/paper/panoflow-learning-optical-flow-for-panoramic","slug":"panoflow-learning-optical-flow-for-panoramic","title":"PanoFlow: Learning 360° Optical Flow for Surrounding Temporal Understanding","date":"2022-02-27","arxiv_id":"2202.13388","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["masterhow/panoflow"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"context-hierarchy-inverse-reinforcement","title":"Context-Hierarchy Inverse Reinforcement Learning","date":"2022-02-25","arxiv_id":"2202.12597","n_code_links":0,"syntology":null},{"paper":null,"slug":"consistent-dropout-for-policy-gradient","title":"Consistent Dropout for Policy Gradient Reinforcement Learning","date":"2022-02-23","arxiv_id":"2202.11818","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-eye-driving-with-the-eyes-of-ai-for-corner","title":"A-Eye: Driving with the Eyes of AI for Corner Case Generation","date":"2022-02-22","arxiv_id":"2202.10803","n_code_links":0,"syntology":null},{"paper":"/paper/don-t-touch-what-matters-task-aware-lipschitz","slug":"don-t-touch-what-matters-task-aware-lipschitz","title":"Don't Touch What Matters: Task-Aware Lipschitz Data Augmentation for Visual Reinforcement Learning","date":"2022-02-21","arxiv_id":"2202.09982","n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-a-shield-from-catastrophic-action","title":"Learning a Shield from Catastrophic Action Effects: Never Repeat the Same Mistake","date":"2022-02-19","arxiv_id":"2202.09516","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-task-safe-reinforcement-learning-for","title":"Multi-task Safe Reinforcement Learning for Navigating Intersections in Dense Traffic","date":"2022-02-19","arxiv_id":"2202.09644","n_code_links":0,"syntology":null},{"paper":"/paper/cadre-a-cascade-deep-reinforcement-learning","slug":"cadre-a-cascade-deep-reinforcement-learning","title":"CADRE: A Cascade Deep Reinforcement Learning Framework for Vision-based Autonomous Urban Driving","date":"2022-02-17","arxiv_id":"2202.08557","n_code_links":1,"syntology":{"ran":8,"of":9,"n_ran_checked":8,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"8 ran (of which 7 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["BIT-MCS/Cadre"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":7,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"omnisyn-synthesizing-360-videos-with-wide","title":"OmniSyn: Synthesizing 360 Videos with Wide-baseline Panoramas","date":"2022-02-17","arxiv_id":"2202.08752","n_code_links":0,"syntology":null},{"paper":"/paper/multi-modal-fusion-for-sensorimotor","slug":"multi-modal-fusion-for-sensorimotor","title":"Multi-Modal Fusion for Sensorimotor Coordination in Steering Angle Prediction","date":"2022-02-11","arxiv_id":"2202.05500","n_code_links":1,"syntology":null},{"paper":"/paper/skrl-modular-and-flexible-library-for","slug":"skrl-modular-and-flexible-library-for","title":"skrl: Modular and Flexible Library for Reinforcement Learning","date":"2022-02-08","arxiv_id":"2202.03825","n_code_links":1,"syntology":null},{"paper":null,"slug":"simulation-to-reality-domain-adaptation-for","title":"Simulation-to-Reality domain adaptation for offline 3D object annotation on pointclouds with correlation alignment","date":"2022-02-06","arxiv_id":"2202.02666","n_code_links":0,"syntology":null},{"paper":null,"slug":"ai-as-a-service-toolkit-for-human-centered","title":"AI-as-a-Service Toolkit for Human-Centered Intelligence in Autonomous Driving","date":"2022-02-03","arxiv_id":"2202.01645","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-machine-learning-smartphone-based-sensing","title":"A Machine Learning Smartphone-based Sensing for Driver Behavior Classification","date":"2022-02-01","arxiv_id":"2202.01893","n_code_links":0,"syntology":null},{"paper":null,"slug":"monotonic-improvement-guarantees-under-non-1","title":"Trust Region Bounds for Decentralized PPO Under Non-stationarity","date":"2022-01-31","arxiv_id":"2202.00082","n_code_links":0,"syntology":null},{"paper":null,"slug":"you-may-not-need-ratio-clipping-in-ppo","title":"You May Not Need Ratio Clipping in PPO","date":"2022-01-31","arxiv_id":"2202.00079","n_code_links":0,"syntology":null},{"paper":null,"slug":"ray-based-distributed-autonomous-vehicle","title":"Ray Based Distributed Autonomous Vehicle Research Platform","date":"2022-01-18","arxiv_id":"2201.06835","n_code_links":0,"syntology":null},{"paper":null,"slug":"spatiotemporal-costmap-inference-for-mpc-via","title":"Spatiotemporal Costmap Inference for MPC via Deep Inverse Reinforcement Learning","date":"2022-01-17","arxiv_id":"2201.06539","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-37-implementation-details-of-proximal","title":"The 37 Implementation Details of Proximal Policy Optimization","date":"2022-01-17","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"a-study-on-mitigating-hard-boundaries-of","title":"A Study on Mitigating Hard Boundaries of Decision-Tree-based Uncertainty Estimates for AI Models","date":"2022-01-10","arxiv_id":"2201.03263","n_code_links":0,"syntology":null},{"paper":"/paper/mirror-learning-a-unifying-framework-of","slug":"mirror-learning-a-unifying-framework-of","title":"Mirror Learning: A Unifying Framework of Policy Optimisation","date":"2022-01-07","arxiv_id":"2201.02373","n_code_links":1,"syntology":null},{"paper":"/paper/dreyevr-democratizing-driving-simulation-in","slug":"dreyevr-democratizing-driving-simulation-in","title":"DReyeVR: Democratizing Virtual Reality Driving Simulation for Behavioural & Interaction Research","date":"2022-01-06","arxiv_id":"2201.01931","n_code_links":2,"syntology":null},{"paper":null,"slug":"towards-robustness-of-neural-networks","title":"Towards Robustness of Neural Networks","date":"2021-12-30","arxiv_id":"2112.15188","n_code_links":0,"syntology":null},{"paper":null,"slug":"ddpg-car-following-model-with-real-world","title":"Modified DDPG car-following model with a real-world human driving experience with CARLA simulator","date":"2021-12-29","arxiv_id":"2112.14602","n_code_links":0,"syntology":null},{"paper":null,"slug":"2112-13937","title":"Multiagent Model-based Credit Assignment for Continuous Control","date":"2021-12-27","arxiv_id":"2112.13937","n_code_links":0,"syntology":null},{"paper":"/paper/intelligent-traffic-light-via-policy-based","slug":"intelligent-traffic-light-via-policy-based","title":"Intelligent Traffic Light via Policy-based Deep Reinforcement Learning","date":"2021-12-27","arxiv_id":"2112.13817","n_code_links":1,"syntology":null},{"paper":null,"slug":"doppler-velocity-based-algorithm-for","title":"Doppler velocity-based algorithm for Clustering and Velocity Estimation of moving objects","date":"2021-12-24","arxiv_id":"2112.12984","n_code_links":0,"syntology":null},{"paper":"/paper/intersection-focused-situation-coverage-based","slug":"intersection-focused-situation-coverage-based","title":"Intersection focused Situation Coverage-based Verification and Validation Framework for Autonomous Vehicles Implemented in CARLA","date":"2021-12-24","arxiv_id":"2112.14706","n_code_links":1,"syntology":null},{"paper":"/paper/maximum-entropy-population-based-training-for-1","slug":"maximum-entropy-population-based-training-for-1","title":"Maximum Entropy Population-Based Training for Zero-Shot Human-AI Coordination","date":"2021-12-22","arxiv_id":"2112.11701","n_code_links":3,"syntology":null},{"paper":null,"slug":"a-deep-reinforcement-learning-model-for","title":"A deep reinforcement learning model for predictive maintenance planning of road assets: Integrating LCA and LCCA","date":"2021-12-20","arxiv_id":"2112.12589","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-reward-machines-a-study-in-partially","title":"Learning Reward Machines: A Study in Partially Observable Reinforcement Learning","date":"2021-12-17","arxiv_id":"2112.09477","n_code_links":0,"syntology":null},{"paper":null,"slug":"scientific-discovery-and-the-cost-of","title":"Scientific Discovery and the Cost of Measurement -- Balancing Information and Cost in Reinforcement Learning","date":"2021-12-14","arxiv_id":"2112.07535","n_code_links":0,"syntology":null},{"paper":"/paper/real-time-collision-risk-estimation-based-on","slug":"real-time-collision-risk-estimation-based-on","title":"Real-time Collision Risk Estimation based on Stochastic Reachability Spaces","date":"2021-12-08","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"ptr-ppo-proximal-policy-optimization-with","title":"PTR-PPO: Proximal Policy Optimization with Prioritized Trajectory Replay","date":"2021-12-07","arxiv_id":"2112.03798","n_code_links":0,"syntology":null},{"paper":null,"slug":"personalized-federated-learning-of-driver","title":"Personalized Federated Learning of Driver Prediction Models for Autonomous Driving","date":"2021-12-02","arxiv_id":"2112.00956","n_code_links":0,"syntology":null},{"paper":null,"slug":"paris-carla-3d-a-real-and-synthetic-outdoor","title":"Paris-CARLA-3D: A Real and Synthetic Outdoor Point Cloud Dataset for Challenging Tasks in 3D Mapping","date":"2021-11-22","arxiv_id":"2111.11348","n_code_links":0,"syntology":null},{"paper":"/paper/learning-robust-output-control-barrier","slug":"learning-robust-output-control-barrier","title":"Learning Robust Output Control Barrier Functions from Safe Expert Demonstrations","date":"2021-11-18","arxiv_id":"2111.09971","n_code_links":1,"syntology":null},{"paper":"/paper/gri-general-reinforced-imitation-and-its","slug":"gri-general-reinforced-imitation-and-its","title":"GRI: General Reinforced Imitation and its Application to Vision-Based Autonomous Driving","date":"2021-11-16","arxiv_id":"2111.08575","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-pseudo-projection-operator-applications","title":"The Pseudo Projection Operator: Applications of Deep Learning to Projection Based Filtering in Non-Trivial Frequency Regimes","date":"2021-11-13","arxiv_id":"2111.07140","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-comparison-of-model-free-and-model","title":"A Comparison of Model-Free and Model Predictive Control for Price Responsive Water Heaters","date":"2021-11-08","arxiv_id":"2111.04689","n_code_links":0,"syntology":null},{"paper":"/paper/coordinated-proximal-policy-optimization","slug":"coordinated-proximal-policy-optimization","title":"Coordinated Proximal Policy Optimization","date":"2021-11-07","arxiv_id":"2111.04051","n_code_links":1,"syntology":null},{"paper":null,"slug":"ai-based-radio-resource-management-and","title":"AI-based Radio Resource Management and Trajectory Design for PD-NOMA Communication in IRS-UAV Assisted Networks","date":"2021-11-06","arxiv_id":"2111.03869","n_code_links":0,"syntology":null},{"paper":null,"slug":"tnd-nas-towards-non-differentiable-objectives","title":"TND-NAS: Towards Non-differentiable Objectives in Progressive Differentiable NAS Framework","date":"2021-11-06","arxiv_id":"2111.03892","n_code_links":0,"syntology":null},{"paper":"/paper/compressing-sensor-data-for-remote-assistance","slug":"compressing-sensor-data-for-remote-assistance","title":"Compressing Sensor Data for Remote Assistance of Autonomous Vehicles using Deep Generative Models","date":"2021-11-05","arxiv_id":"2111.03201","n_code_links":1,"syntology":null},{"paper":null,"slug":"improving-rna-secondary-structure-design","title":"Improving RNA Secondary Structure Design using Deep Reinforcement Learning","date":"2021-11-05","arxiv_id":"2111.04504","n_code_links":0,"syntology":null},{"paper":"/paper/learning-distilled-collaboration-graph-for","slug":"learning-distilled-collaboration-graph-for","title":"Learning Distilled Collaboration Graph for Multi-Agent Perception","date":"2021-11-01","arxiv_id":"2111.00643","n_code_links":2,"syntology":{"ran":4,"of":6,"n_ran_checked":4,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","official":{"repos":["ai4ce/DiscoNet"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/object-aware-regularization-for-addressing","slug":"object-aware-regularization-for-addressing","title":"Object-Aware Regularization for Addressing Causal Confusion in Imitation Learning","date":"2021-10-27","arxiv_id":"2110.14118","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["alinlab/oreo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"a-deep-reinforcement-learning-approach-for-9","title":"A Deep Reinforcement Learning Approach for Audio-based Navigation and Audio Source Localization in Multi-speaker Environments","date":"2021-10-25","arxiv_id":"2110.12778","n_code_links":0,"syntology":null},{"paper":null,"slug":"cim-ppo-proximal-policy-optimization-with-liu","title":"CIM-PPO:Proximal Policy Optimization with Liu-Correntropy Induced Metric","date":"2021-10-20","arxiv_id":"2110.10522","n_code_links":0,"syntology":null},{"paper":null,"slug":"generative-adversarial-imitation-learning-for-1","title":"Generative Adversarial Imitation Learning for End-to-End Autonomous Driving on Urban Environments","date":"2021-10-16","arxiv_id":"2110.08586","n_code_links":0,"syntology":null},{"paper":"/paper/improving-the-sample-efficiency-of-neural","slug":"improving-the-sample-efficiency-of-neural","title":"Improving the sample-efficiency of neural architecture search with reinforcement learning","date":"2021-10-13","arxiv_id":"2110.06751","n_code_links":1,"syntology":null},{"paper":null,"slug":"navigation-in-urban-environments-amongst","title":"Navigation In Urban Environments Amongst Pedestrians Using Multi-Objective Deep Reinforcement Learning","date":"2021-10-11","arxiv_id":"2110.05205","n_code_links":0,"syntology":null},{"paper":"/paper/camera-calibration-through-camera-projection","slug":"camera-calibration-through-camera-projection","title":"Camera Calibration through Camera Projection Loss","date":"2021-10-07","arxiv_id":"2110.03479","n_code_links":2,"syntology":null},{"paper":null,"slug":"cycle-consistent-world-models-for-domain","title":"Cycle-Consistent World Models for Domain Independent Latent Imagination","date":"2021-10-02","arxiv_id":"2110.00808","n_code_links":0,"syntology":null},{"paper":null,"slug":"bitcoin-transaction-strategy-construction","title":"Bitcoin Transaction Strategy Construction Based on Deep Reinforcement Learning","date":"2021-09-30","arxiv_id":"2109.14789","n_code_links":0,"syntology":null},{"paper":null,"slug":"fight-fire-with-fire-countering-bad-shortcuts","title":"Fight fire with fire: countering bad shortcuts in imitation learning with good shortcuts","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null}],"record_sha256":"d1c090c38d9f4cc9e98ffd7d1ea0d5aca9bcf0b6dc34a267c9d001c661992894","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}