{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/ppo/papers/6","list_of":"/method/ppo","method":"PPO","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":6,"pages_in_order":10,"rows_per_page":100,"rows":[501,600],"of":949,"counts":{"archive_papers_tagged":949,"with_a_code_link":397,"where_syntology_ran_a_sample":139,"not_listed_spam_title":0,"listed":949,"listed_where_code_ran":139,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":114,"every_run_a_failure_of_syntologys_instrument":25,"listed_with_a_run_with_no_instrument_failure":114,"listed_every_run_a_failure_of_syntologys_instrument":25,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/ppo","prev":"/method/ppo/papers/5","next":"/method/ppo/papers/7","papers":[{"paper":null,"slug":"implicit-ray-transformers-for-multi-view","title":"Implicit Ray-Transformers for Multi-view Remote Sensing Image Segmentation","date":"2023-03-15","arxiv_id":"2303.08401","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-based-wavefront","title":"Reinforcement Learning-based Wavefront Sensorless Adaptive Optics Approaches for Satellite-to-Ground Laser Communication","date":"2023-03-13","arxiv_id":"2303.07516","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-strategy-oriented-bayesian-soft-actor","title":"A Strategy-Oriented Bayesian Soft Actor-Critic Model","date":"2023-03-07","arxiv_id":"2303.04193","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-reinforcement-learning-approach-for-7","title":"A Reinforcement Learning Approach for Scheduling Problems With Improved Generalization Through Order Swapping","date":"2023-02-27","arxiv_id":"2302.13941","n_code_links":0,"syntology":null},{"paper":null,"slug":"seo-safety-aware-energy-optimization","title":"SEO: Safety-Aware Energy Optimization Framework for Multi-Sensor Neural Controllers at the Edge","date":"2023-02-24","arxiv_id":"2302.12493","n_code_links":0,"syntology":null},{"paper":"/paper/behavior-proximal-policy-optimization","slug":"behavior-proximal-policy-optimization","title":"Behavior Proximal Policy Optimization","date":"2023-02-22","arxiv_id":"2302.11312","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["dragon-zhuang/bppo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/dynamic-simplex-balancing-safety-and","slug":"dynamic-simplex-balancing-safety-and","title":"Dynamic Simplex: Balancing Safety and Performance in Autonomous Cyber Physical Systems","date":"2023-02-20","arxiv_id":"2302.09750","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["BaitingLuo/Dynamic_Simplex"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"towards-co-operative-congestion-mitigation","title":"Towards Co-operative Congestion Mitigation","date":"2023-02-17","arxiv_id":"2302.09140","n_code_links":0,"syntology":null},{"paper":null,"slug":"cooperative-perception-for-safe-control-of","title":"Cooperative Perception for Safe Control of Autonomous Vehicles under LiDAR Spoofing Attacks","date":"2023-02-14","arxiv_id":"2302.07341","n_code_links":0,"syntology":null},{"paper":null,"slug":"energyshield-provably-safe-offloading-of","title":"EnergyShield: Provably-Safe Offloading of Neural Network Controllers for Energy Efficiency","date":"2023-02-13","arxiv_id":"2302.06572","n_code_links":0,"syntology":null},{"paper":null,"slug":"shared-information-based-safe-and-efficient","title":"Shared Information-Based Safe And Efficient Behavior Planning For Connected Autonomous Vehicles","date":"2023-02-08","arxiv_id":"2302.04321","n_code_links":0,"syntology":null},{"paper":"/paper/sample-dropout-a-simple-yet-effective","slug":"sample-dropout-a-simple-yet-effective","title":"Sample Dropout: A Simple yet Effective Variance Reduction Technique in Deep Policy Optimization","date":"2023-02-05","arxiv_id":"2302.02299","n_code_links":1,"syntology":{"ran":6,"of":12,"n_ran_checked":5,"n_instrument":1,"unverified":6,"pointer_only":12,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","official":{"repos":["linzichuan/sdpo"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/steps-joint-self-supervised-nighttime-image","slug":"steps-joint-self-supervised-nighttime-image","title":"STEPS: Joint Self-supervised Nighttime Image Enhancement and Depth Estimation","date":"2023-02-02","arxiv_id":"2302.01334","n_code_links":1,"syntology":null},{"paper":null,"slug":"bridging-physics-informed-neural-networks","title":"Bridging Physics-Informed Neural Networks with Reinforcement Learning: Hamilton-Jacobi-Bellman Proximal Policy Optimization (HJBPPO)","date":"2023-02-01","arxiv_id":"2302.00237","n_code_links":0,"syntology":null},{"paper":"/paper/learning-fast-and-slow-a-goal-directed-memory","slug":"learning-fast-and-slow-a-goal-directed-memory","title":"Learning, Fast and Slow: A Goal-Directed Memory-Based Approach for Dynamic Environments","date":"2023-01-31","arxiv_id":"2301.13758","n_code_links":1,"syntology":null},{"paper":"/paper/a-novel-framework-for-policy-mirror-descent","slug":"a-novel-framework-for-policy-mirror-descent","title":"A Novel Framework for Policy Mirror Descent with General Parameterization and Linear Convergence","date":"2023-01-30","arxiv_id":"2301.13139","n_code_links":1,"syntology":null},{"paper":null,"slug":"incorporating-recurrent-reinforcement","title":"Incorporating Recurrent Reinforcement Learning into Model Predictive Control for Adaptive Control in Autonomous Driving","date":"2023-01-30","arxiv_id":"2301.13313","n_code_links":0,"syntology":null},{"paper":null,"slug":"softtreemax-exponential-variance-reduction-in","title":"SoftTreeMax: Exponential Variance Reduction in Policy Gradient via Tree Search","date":"2023-01-30","arxiv_id":"2301.13236","n_code_links":0,"syntology":null},{"paper":"/paper/joint-action-loss-for-proximal-policy","slug":"joint-action-loss-for-proximal-policy","title":"Joint action loss for proximal policy optimization","date":"2023-01-26","arxiv_id":"2301.10919","n_code_links":1,"syntology":null},{"paper":"/paper/schlably-a-python-framework-for-deep","slug":"schlably-a-python-framework-for-deep","title":"schlably: A Python Framework for Deep Reinforcement Learning Based Scheduling Experiments","date":"2023-01-10","arxiv_id":"2301.04182","n_code_links":1,"syntology":null},{"paper":"/paper/asynchronous-multi-agent-reinforcement","slug":"asynchronous-multi-agent-reinforcement","title":"Asynchronous Multi-Agent Reinforcement Learning for Efficient Real-Time Multi-Robot Cooperative Exploration","date":"2023-01-09","arxiv_id":"2301.03398","n_code_links":2,"syntology":null},{"paper":null,"slug":"tuning-path-tracking-controllers-for","title":"Tuning Path Tracking Controllers for Autonomous Cars Using Reinforcement Learning","date":"2023-01-09","arxiv_id":"2301.03363","n_code_links":0,"syntology":null},{"paper":null,"slug":"e-inu-simulating-a-quadruped-robot-with","title":"e-Inu: Simulating A Quadruped Robot With Emotional Sentience","date":"2023-01-03","arxiv_id":"2301.00964","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-for-asset-1","title":"Deep Reinforcement Learning for Asset Allocation: Reward Clipping","date":"2023-01-02","arxiv_id":"2301.05300","n_code_links":0,"syntology":null},{"paper":null,"slug":"simoun-synergizing-interactive-motion","title":"Simoun: Synergizing Interactive Motion-appearance Understanding for Vision-based Reinforcement Learning","date":"2023-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"stabilizing-visual-reinforcement-learning-via","title":"Stabilizing Visual Reinforcement Learning via Asymmetric Interactive Cooperation","date":"2023-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-automating-codenames-spymasters-with","title":"Towards automating Codenames spymasters with deep reinforcement learning","date":"2022-12-28","arxiv_id":"2212.14104","n_code_links":0,"syntology":null},{"paper":null,"slug":"hierarchical-deep-reinforcement-learning-for","title":"Hierarchical Deep Reinforcement Learning for Age-of-Information Minimization in IRS-aided and Wireless-powered Wireless Networks","date":"2022-12-27","arxiv_id":"2212.13390","n_code_links":0,"syntology":null},{"paper":"/paper/learning-generalizable-representations-for-1","slug":"learning-generalizable-representations-for-1","title":"Learning Generalizable Representations for Reinforcement Learning via Adaptive Meta-learner of Behavioral Similarities","date":"2022-12-26","arxiv_id":"2212.13088","n_code_links":1,"syntology":null},{"paper":"/paper/lifelong-reinforcement-learning-with","slug":"lifelong-reinforcement-learning-with","title":"Lifelong Reinforcement Learning with Modulating Masks","date":"2022-12-21","arxiv_id":"2212.11110","n_code_links":4,"syntology":null},{"paper":null,"slug":"pre-trained-image-encoder-for-generalizable","title":"Pre-Trained Image Encoder for Generalizable Visual Reinforcement Learning","date":"2022-12-17","arxiv_id":"2212.08860","n_code_links":0,"syntology":null},{"paper":null,"slug":"distribution-aware-goal-prediction-and","title":"Distribution-aware Goal Prediction and Conformant Model-based Planning for Safe Autonomous Driving","date":"2022-12-16","arxiv_id":"2212.08729","n_code_links":0,"syntology":null},{"paper":"/paper/learning-for-vehicle-to-vehicle-cooperative","slug":"learning-for-vehicle-to-vehicle-cooperative","title":"Learning for Vehicle-to-Vehicle Cooperative Perception under Lossy Communication","date":"2022-12-16","arxiv_id":"2212.08273","n_code_links":1,"syntology":null},{"paper":null,"slug":"multi-agent-reinforcement-learning-with-4","title":"Multi-Agent Reinforcement Learning with Shared Resources for Inventory Management","date":"2022-12-15","arxiv_id":"2212.07684","n_code_links":0,"syntology":null},{"paper":"/paper/robust-policy-optimization-in-deep","slug":"robust-policy-optimization-in-deep","title":"Robust Policy Optimization in Deep Reinforcement Learning","date":"2022-12-14","arxiv_id":"2212.07536","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["vwxyzjn/cleanrl"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"paper":null,"slug":"ppo-ue-proximal-policy-optimization-via","title":"PPO-UE: Proximal Policy Optimization via Uncertainty-Aware Exploration","date":"2022-12-13","arxiv_id":"2212.06343","n_code_links":0,"syntology":null},{"paper":null,"slug":"decentralized-cooperative-perception-for","title":"Decentralized cooperative perception for autonomous vehicles: Learning to value the unknown","date":"2022-12-12","arxiv_id":"2301.01250","n_code_links":0,"syntology":null},{"paper":"/paper/solving-the-side-chain-packing-arrangement-of","slug":"solving-the-side-chain-packing-arrangement-of","title":"Reinforcement Learning for Molecular Dynamics Optimization: A Stochastic Pontryagin Maximum Principle Approach","date":"2022-12-06","arxiv_id":"2212.03320","n_code_links":1,"syntology":null},{"paper":null,"slug":"safe-reinforcement-learning-with","title":"Safe Reinforcement Learning with Probabilistic Control Barrier Functions for Ramp Merging","date":"2022-12-01","arxiv_id":"2212.00618","n_code_links":0,"syntology":null},{"paper":"/paper/analyzing-infrastructure-lidar-placement-with","slug":"analyzing-infrastructure-lidar-placement-with","title":"Analyzing Infrastructure LiDAR Placement with Realistic LiDAR Simulation Library","date":"2022-11-29","arxiv_id":"2211.15975","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["pjlab-adg/lidarsimlib-and-placement-evaluation","pjlab-adg/pcsim"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"combined-peak-reduction-and-self-consumption","title":"Combined Peak Reduction and Self-Consumption Using Proximal Policy Optimization","date":"2022-11-27","arxiv_id":"2211.14831","n_code_links":0,"syntology":null},{"paper":"/paper/multi-task-learning-for-camera-calibration","slug":"multi-task-learning-for-camera-calibration","title":"Multi-task Learning for Camera Calibration","date":"2022-11-22","arxiv_id":"2211.12432","n_code_links":2,"syntology":null},{"paper":null,"slug":"rationale-aware-autonomous-driving-policy","title":"Rationale-aware Autonomous Driving Policy utilizing Safety Force Field implemented on CARLA Simulator","date":"2022-11-18","arxiv_id":"2211.10237","n_code_links":0,"syntology":null},{"paper":"/paper/dynamic-conditional-imitation-learning-for-1","slug":"dynamic-conditional-imitation-learning-for-1","title":"Dynamic Conditional Imitation Learning for Autonomous Driving","date":"2022-11-17","arxiv_id":"2211.11579","n_code_links":1,"syntology":null},{"paper":"/paper/efficient-deep-reinforcement-learning-with-1","slug":"efficient-deep-reinforcement-learning-with-1","title":"Efficient Deep Reinforcement Learning with Predictive Processing Proximal Policy Optimization","date":"2022-11-11","arxiv_id":"2211.06236","n_code_links":1,"syntology":null},{"paper":null,"slug":"estimation-of-appearance-and-occupancy","title":"Estimation of Appearance and Occupancy Information in Birds Eye View from Surround Monocular Images","date":"2022-11-08","arxiv_id":"2211.04557","n_code_links":0,"syntology":null},{"paper":null,"slug":"decentralized-policy-optimization","title":"Decentralized Policy Optimization","date":"2022-11-06","arxiv_id":"2211.03032","n_code_links":0,"syntology":null},{"paper":"/paper/design-process-is-a-reinforcement-learning","slug":"design-process-is-a-reinforcement-learning","title":"Design Process is a Reinforcement Learning Problem","date":"2022-11-06","arxiv_id":"2211.03136","n_code_links":1,"syntology":null},{"paper":"/paper/defix-detecting-and-fixing-failure-scenarios","slug":"defix-detecting-and-fixing-failure-scenarios","title":"DeFIX: Detecting and Fixing Failure Scenarios with Reinforcement Learning in Imitation Learning Based Autonomous Driving","date":"2022-10-29","arxiv_id":"2210.16567","n_code_links":2,"syntology":null},{"paper":"/paper/self-improving-safety-performance-of","slug":"self-improving-safety-performance-of","title":"Self-Improving Safety Performance of Reinforcement Learning Based Driving with Black-Box Verification Algorithms","date":"2022-10-29","arxiv_id":"2210.16575","n_code_links":2,"syntology":null},{"paper":null,"slug":"many-objective-reinforcement-learning-for","title":"Many-Objective Reinforcement Learning for Online Testing of DNN-Enabled Systems","date":"2022-10-27","arxiv_id":"2210.15432","n_code_links":0,"syntology":null},{"paper":"/paper/plant-explainable-planning-transformers-via","slug":"plant-explainable-planning-transformers-via","title":"PlanT: Explainable Planning Transformers via Object-Level Representations","date":"2022-10-25","arxiv_id":"2210.14222","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["autonomousvision/plant"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"out-of-distribution-reasoning-by-weakly","title":"Out of Distribution Reasoning by Weakly-Supervised Disentangled Logic Variational Autoencoder","date":"2022-10-18","arxiv_id":"2210.09959","n_code_links":0,"syntology":null},{"paper":"/paper/a-multilevel-reinforcement-learning-framework","slug":"a-multilevel-reinforcement-learning-framework","title":"A Multilevel Reinforcement Learning Framework for PDE-based Control","date":"2022-10-15","arxiv_id":"2210.08400","n_code_links":2,"syntology":null},{"paper":"/paper/model-based-imitation-learning-for-urban","slug":"model-based-imitation-learning-for-urban","title":"Model-Based Imitation Learning for Urban Driving","date":"2022-10-14","arxiv_id":"2210.07729","n_code_links":1,"syntology":{"ran":21,"of":23,"n_ran_checked":13,"n_instrument":8,"unverified":2,"pointer_only":0,"phrase":"21 ran (of which 13 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 8 where Syntology's instrument failed) · 2 unverified","official":{"repos":["wayveai/mile"],"state":"official (archive's flag): 21 ran","n_ran":21,"n_constructed":13,"n_ran_no_instrument_failure":13,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"learning-driving-policies-for-end-to-end","title":"Exploring Contextual Representation and Multi-Modality for End-to-End Autonomous Driving","date":"2022-10-13","arxiv_id":"2210.06758","n_code_links":0,"syntology":null},{"paper":"/paper/discovered-policy-optimisation","slug":"discovered-policy-optimisation","title":"Discovered Policy Optimisation","date":"2022-10-11","arxiv_id":"2210.05639","n_code_links":1,"syntology":null},{"paper":null,"slug":"enhance-sample-efficiency-and-robustness-of","title":"Enhance Sample Efficiency and Robustness of End-to-end Urban Autonomous Driving via Semantic Masked World Model","date":"2022-10-08","arxiv_id":"2210.04017","n_code_links":0,"syntology":null},{"paper":"/paper/real-time-reinforcement-learning-for-vision","slug":"real-time-reinforcement-learning-for-vision","title":"Real-Time Reinforcement Learning for Vision-Based Robotics Utilizing Local and Remote Computers","date":"2022-10-05","arxiv_id":"2210.02317","n_code_links":2,"syntology":{"ran":4,"of":4,"n_ran_checked":2,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["rlai-lab/relod","rlai-lab/remote-onboard-agent"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"spatial-temporal-aware-safe-multi-agent","title":"Spatial-Temporal-Aware Safe Multi-Agent Reinforcement Learning of Connected Autonomous Vehicles in Challenging Scenarios","date":"2022-10-05","arxiv_id":"2210.02300","n_code_links":0,"syntology":null},{"paper":"/paper/is-reinforcement-learning-not-for-natural","slug":"is-reinforcement-learning-not-for-natural","title":"Is Reinforcement Learning (Not) for Natural Language Processing: Benchmarks, Baselines, and Building Blocks for Natural Language Policy Optimization","date":"2022-10-03","arxiv_id":"2210.01241","n_code_links":3,"syntology":null},{"paper":null,"slug":"multi-agent-chance-constrained-stochastic","title":"Multi-Agent Chance-Constrained Stochastic Shortest Path with Application to Risk-Aware Intelligent Intersection","date":"2022-10-03","arxiv_id":"2210.01766","n_code_links":0,"syntology":null},{"paper":null,"slug":"obstacle-avoidance-for-robotic-manipulator-in","title":"IPPO: Obstacle Avoidance for Robotic Manipulators in Joint Space via Improved Proximal Policy Optimization","date":"2022-10-03","arxiv_id":"2210.00803","n_code_links":0,"syntology":null},{"paper":null,"slug":"softtreemax-policy-gradient-with-tree-search","title":"SoftTreeMax: Policy Gradient with Tree Search","date":"2022-09-28","arxiv_id":"2209.13966","n_code_links":0,"syntology":null},{"paper":null,"slug":"lamarckian-platform-pushing-the-boundaries-of","title":"Lamarckian Platform: Pushing the Boundaries of Evolutionary Reinforcement Learning towards Asynchronous Commercial Games","date":"2022-09-21","arxiv_id":"2209.10055","n_code_links":0,"syntology":null},{"paper":null,"slug":"model-free-reinforcement-learning-for-asset","title":"Model-Free Reinforcement Learning for Asset Allocation","date":"2022-09-21","arxiv_id":"2209.10458","n_code_links":0,"syntology":null},{"paper":"/paper/partial-observability-during-drl-for-robot","slug":"partial-observability-during-drl-for-robot","title":"Experimental Study on The Effect of Multi-step Deep Reinforcement Learning in POMDPs","date":"2022-09-12","arxiv_id":"2209.04999","n_code_links":1,"syntology":null},{"paper":null,"slug":"normality-guided-distributional-reinforcement","title":"Normality-Guided Distributional Reinforcement Learning for Continuous Control","date":"2022-08-28","arxiv_id":"2208.13125","n_code_links":0,"syntology":null},{"paper":null,"slug":"entropy-augmented-reinforcement-learning","title":"Entropy Augmented Reinforcement Learning","date":"2022-08-19","arxiv_id":"2208.09322","n_code_links":0,"syntology":null},{"paper":null,"slug":"path-planning-of-cleaning-robot-with","title":"Path Planning of Cleaning Robot with Reinforcement Learning","date":"2022-08-17","arxiv_id":"2208.08211","n_code_links":0,"syntology":null},{"paper":"/paper/bsac-bayesian-strategy-network-based-soft","slug":"bsac-bayesian-strategy-network-based-soft","title":"Bayesian Soft Actor-Critic: A Directed Acyclic Strategy Graph Based Deep Reinforcement Learning","date":"2022-08-11","arxiv_id":"2208.06033","n_code_links":2,"syntology":null},{"paper":"/paper/aerial-monocular-3d-object-detection","slug":"aerial-monocular-3d-object-detection","title":"Aerial Monocular 3D Object Detection","date":"2022-08-08","arxiv_id":"2208.03974","n_code_links":1,"syntology":null},{"paper":"/paper/learning-to-generalize-with-object-centric","slug":"learning-to-generalize-with-object-centric","title":"Learning to Generalize with Object-centric Agents in the Open World Survival Game Crafter","date":"2022-08-05","arxiv_id":"2208.03374","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":3,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["astanic/crafter-ood"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/performance-comparison-of-deep-rl-algorithms","slug":"performance-comparison-of-deep-rl-algorithms","title":"Performance Comparison of Deep RL Algorithms for Energy Systems Optimal Scheduling","date":"2022-08-01","arxiv_id":"2208.00728","n_code_links":1,"syntology":null},{"paper":null,"slug":"adaptive-feature-fusion-for-cooperative","title":"Adaptive Feature Fusion for Cooperative Perception using LiDAR Point Clouds","date":"2022-07-30","arxiv_id":"2208.00116","n_code_links":0,"syntology":null},{"paper":null,"slug":"solving-the-vehicle-routing-problem-with-deep","title":"Solving the vehicle routing problem with deep reinforcement learning","date":"2022-07-30","arxiv_id":"2208.00202","n_code_links":0,"syntology":null},{"paper":"/paper/safety-enhanced-autonomous-driving-using-1","slug":"safety-enhanced-autonomous-driving-using-1","title":"Safety-Enhanced Autonomous Driving Using Interpretable Sensor Fusion Transformer","date":"2022-07-28","arxiv_id":"2207.14024","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["opendilab/InterFuser"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"adaptive-decision-making-at-the-intersection","title":"Adaptive Decision Making at the Intersection for Autonomous Vehicles Based on Skill Discovery","date":"2022-07-24","arxiv_id":"2207.11724","n_code_links":0,"syntology":null},{"paper":"/paper/synthetic-dataset-generation-for-adversarial","slug":"synthetic-dataset-generation-for-adversarial","title":"Synthetic Dataset Generation for Adversarial Machine Learning Research","date":"2022-07-21","arxiv_id":"2207.10719","n_code_links":1,"syntology":null},{"paper":null,"slug":"resolving-copycat-problems-in-visual","title":"Resolving Copycat Problems in Visual Imitation Learning via Residual Action Prediction","date":"2022-07-20","arxiv_id":"2207.09705","n_code_links":0,"syntology":null},{"paper":"/paper/anti-carla-an-adversarial-testing-framework","slug":"anti-carla-an-adversarial-testing-framework","title":"ANTI-CARLA: An Adversarial Testing Framework for Autonomous Vehicles in CARLA","date":"2022-07-19","arxiv_id":"2208.06309","n_code_links":1,"syntology":null},{"paper":"/paper/st-p3-end-to-end-vision-based-autonomous","slug":"st-p3-end-to-end-vision-based-autonomous","title":"ST-P3: End-to-end Vision-based Autonomous Driving via Spatial-Temporal Feature Learning","date":"2022-07-15","arxiv_id":"2207.07601","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":4,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/automated-detection-of-label-errors-in","slug":"automated-detection-of-label-errors-in","title":"Automated Detection of Label Errors in Semantic Segmentation Datasets via Deep Learning and Uncertainty Quantification","date":"2022-07-13","arxiv_id":"2207.06104","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["mrcoee/automatic-label-error-detection"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"towards-global-optimality-in-cooperative-marl","title":"Towards Global Optimality in Cooperative MARL with the Transformation And Distillation Framework","date":"2022-07-12","arxiv_id":"2207.11143","n_code_links":0,"syntology":null},{"paper":"/paper/keep-your-distance-determining-sampling-and","slug":"keep-your-distance-determining-sampling-and","title":"Keep your Distance: Determining Sampling and Distance Thresholds in Machine Learning Monitoring","date":"2022-07-11","arxiv_id":"2207.05078","n_code_links":1,"syntology":null},{"paper":null,"slug":"game-state-learning-via-game-scene","title":"Game State Learning via Game Scene Augmentation","date":"2022-07-04","arxiv_id":"2207.01289","n_code_links":0,"syntology":null},{"paper":null,"slug":"data-generation-using-simulation-technology","title":"Data generation using simulation technology to improve perception mechanism of autonomous vehicles","date":"2022-07-01","arxiv_id":"2207.00191","n_code_links":0,"syntology":null},{"paper":"/paper/learning-mixture-of-domain-specific-experts","slug":"learning-mixture-of-domain-specific-experts","title":"Learning mixture of domain-specific experts via disentangled factors for autonomous driving","date":"2022-06-28","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/ibiscape-a-simulated-benchmark-for-multi","slug":"ibiscape-a-simulated-benchmark-for-multi","title":"IBISCape: A Simulated Benchmark for multi-modal SLAM Systems Evaluation in Large-scale Dynamic Environments","date":"2022-06-27","arxiv_id":"2206.13455","n_code_links":1,"syntology":null},{"paper":null,"slug":"fighting-fire-with-fire-avoiding-dnn","title":"Fighting Fire with Fire: Avoiding DNN Shortcuts through Priming","date":"2022-06-22","arxiv_id":"2206.10816","n_code_links":0,"syntology":null},{"paper":"/paper/multi-agent-car-parking-using-reinforcement","slug":"multi-agent-car-parking-using-reinforcement","title":"Multi-Agent Car Parking using Reinforcement Learning","date":"2022-06-22","arxiv_id":"2206.13338","n_code_links":1,"syntology":null},{"paper":"/paper/envpool-a-highly-parallel-reinforcement","slug":"envpool-a-highly-parallel-reinforcement","title":"EnvPool: A Highly Parallel Reinforcement Learning Environment Execution Engine","date":"2022-06-21","arxiv_id":"2206.10558","n_code_links":3,"syntology":{"ran":5,"of":5,"n_ran_checked":4,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sail-sg/envpool"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["named_in_paper","official"]}}},{"paper":null,"slug":"imitate-then-transcend-multi-agent-optimal","title":"Imitate then Transcend: Multi-Agent Optimal Execution with Dual-Window Denoise PPO","date":"2022-06-21","arxiv_id":"2206.10736","n_code_links":0,"syntology":null},{"paper":null,"slug":"incorporating-voice-instructions-in-model","title":"Incorporating Voice Instructions in Model-Based Reinforcement Learning for Self-Driving Cars","date":"2022-06-21","arxiv_id":"2206.10249","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-parametric-class-of-approximate-gradient","title":"A Parametric Class of Approximate Gradient Updates for Policy Optimization","date":"2022-06-17","arxiv_id":"2206.08499","n_code_links":0,"syntology":null},{"paper":"/paper/towards-human-level-bimanual-dexterous","slug":"towards-human-level-bimanual-dexterous","title":"Towards Human-Level Bimanual Dexterous Manipulation with Reinforcement Learning","date":"2022-06-17","arxiv_id":"2206.08686","n_code_links":1,"syntology":null},{"paper":null,"slug":"level-2-autonomous-driving-on-a-single-device","title":"Level 2 Autonomous Driving on a Single Device: Diving into the Devils of Openpilot","date":"2022-06-16","arxiv_id":"2206.08176","n_code_links":0,"syntology":null},{"paper":"/paper/trajectory-guided-control-prediction-for-end","slug":"trajectory-guided-control-prediction-for-end","title":"Trajectory-guided Control Prediction for End-to-end Autonomous Driving: A Simple yet Strong Baseline","date":"2022-06-16","arxiv_id":"2206.08129","n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-task-independent-game-state","title":"Learning Task-Independent Game State Representations from Unlabeled Images","date":"2022-06-13","arxiv_id":"2206.06490","n_code_links":0,"syntology":null},{"paper":"/paper/a-unified-approach-to-reinforcement-learning","slug":"a-unified-approach-to-reinforcement-learning","title":"A Unified Approach to Reinforcement Learning, Quantal Response Equilibria, and Two-Player Zero-Sum Games","date":"2022-06-12","arxiv_id":"2206.05825","n_code_links":3,"syntology":{"ran":5,"of":5,"n_ran_checked":4,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 3 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["deepmind/open_spiel"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["named_in_paper"]}}}],"record_sha256":"5fcdfe2c6ec9d136a38ef076d73160f19f28c68d64754cb62c9fea358f0f96fb","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}