{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/ppo/papers/2","list_of":"/method/ppo","method":"PPO","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":2,"pages_in_order":10,"rows_per_page":100,"rows":[101,200],"of":949,"counts":{"archive_papers_tagged":949,"with_a_code_link":397,"where_syntology_ran_a_sample":139,"not_listed_spam_title":0,"listed":949,"listed_where_code_ran":139,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":114,"every_run_a_failure_of_syntologys_instrument":25,"listed_with_a_run_with_no_instrument_failure":114,"listed_every_run_a_failure_of_syntologys_instrument":25,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/ppo","prev":"/method/ppo","next":"/method/ppo/papers/3","papers":[{"paper":"/paper/a-comprehensive-llm-powered-framework-for","slug":"a-comprehensive-llm-powered-framework-for","title":"A Comprehensive LLM-powered Framework for Driving Intelligence Evaluation","date":"2025-03-07","arxiv_id":"2503.05164","n_code_links":2,"syntology":null},{"paper":"/paper/bevdriver-leveraging-bev-maps-in-llms-for","slug":"bevdriver-leveraging-bev-maps-in-llms-for","title":"BEVDriver: Leveraging BEV Maps in LLMs for Robust Closed-Loop Driving","date":"2025-03-05","arxiv_id":"2503.03074","n_code_links":1,"syntology":null},{"paper":null,"slug":"2503-01676","title":"Perceptual Motor Learning with Active Inference Framework for Robust Lateral Control","date":"2025-03-03","arxiv_id":"2503.01676","n_code_links":0,"syntology":null},{"paper":null,"slug":"caps-context-aware-priority-sampling-for","title":"CAPS: Context-Aware Priority Sampling for Enhanced Imitation Learning in Autonomous Driving","date":"2025-03-03","arxiv_id":"2503.01650","n_code_links":0,"syntology":null},{"paper":null,"slug":"what-s-behind-ppo-s-collapse-in-long-cot","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","date":"2025-03-03","arxiv_id":"2503.01491","n_code_links":0,"syntology":null},{"paper":"/paper/caril-confidence-aware-regression-in","slug":"caril-confidence-aware-regression-in","title":"CARIL: Confidence-Aware Regression in Imitation Learning for Autonomous Driving","date":"2025-03-02","arxiv_id":"2503.00783","n_code_links":1,"syntology":null},{"paper":null,"slug":"multimodal-dreaming-a-global-workspace","title":"Multimodal Dreaming: A Global Workspace Approach to World Model-Based Reinforcement Learning","date":"2025-02-28","arxiv_id":"2502.21142","n_code_links":0,"syntology":null},{"paper":null,"slug":"shared-autonomy-for-proximal-teaching","title":"Shared Autonomy for Proximal Teaching","date":"2025-02-27","arxiv_id":"2502.19899","n_code_links":0,"syntology":null},{"paper":null,"slug":"sorft-issue-resolving-with-subtask-oriented","title":"SoRFT: Issue Resolving with Subtask-oriented Reinforced Fine-Tuning","date":"2025-02-27","arxiv_id":"2502.20127","n_code_links":0,"syntology":null},{"paper":null,"slug":"lean-and-mean-decoupled-value-policy","title":"Lean and Mean: Decoupled Value Policy Optimization with Global Value Guidance","date":"2025-02-24","arxiv_id":"2502.16944","n_code_links":0,"syntology":null},{"paper":null,"slug":"ensemble-rl-through-classifier-models","title":"Ensemble RL through Classifier Models: Enhancing Risk-Return Trade-offs in Trading Strategies","date":"2025-02-23","arxiv_id":"2502.17518","n_code_links":0,"syntology":null},{"paper":null,"slug":"sprig-stackelberg-perception-reinforcement","title":"SPRIG: Stackelberg Perception-Reinforcement Learning with Internal Game Dynamics","date":"2025-02-20","arxiv_id":"2502.14264","n_code_links":0,"syntology":null},{"paper":"/paper/synth-it-like-kitti-synthetic-data-generation","slug":"synth-it-like-kitti-synthetic-data-generation","title":"Synth It Like KITTI: Synthetic Data Generation for Object Detection in Driving Scenarios","date":"2025-02-20","arxiv_id":"2502.15076","n_code_links":1,"syntology":null},{"paper":null,"slug":"sce2drivex-a-generalized-mllm-framework-for","title":"Sce2DriveX: A Generalized MLLM Framework for Scene-to-Drive Learning","date":"2025-02-19","arxiv_id":"2502.14917","n_code_links":0,"syntology":null},{"paper":null,"slug":"hovering-flight-of-soft-actuated-insect-scale","title":"Hovering Flight of Soft-Actuated Insect-Scale Micro Aerial Vehicles using Deep Reinforcement Learning","date":"2025-02-17","arxiv_id":"2502.12355","n_code_links":0,"syntology":null},{"paper":null,"slug":"mc-bevro-multi-camera-bird-eye-view-road","title":"MC-BEVRO: Multi-Camera Bird Eye View Road Occupancy Detection for Traffic Monitoring","date":"2025-02-16","arxiv_id":"2502.11287","n_code_links":0,"syntology":null},{"paper":"/paper/reevaluating-policy-gradient-methods-for","slug":"reevaluating-policy-gradient-methods-for","title":"Reevaluating Policy Gradient Methods for Imperfect-Information Games","date":"2025-02-13","arxiv_id":"2502.08938","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["gabrfarina/exp-a-spiel","nathanlct/iig-rl-benchmark"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":null,"slug":"salience-invariant-consistent-policy-learning","title":"Salience-Invariant Consistent Policy Learning for Generalization in Visual Reinforcement Learning","date":"2025-02-12","arxiv_id":"2502.08336","n_code_links":0,"syntology":null},{"paper":"/paper/on-the-emergence-of-thinking-in-llms-i","slug":"on-the-emergence-of-thinking-in-llms-i","title":"On the Emergence of Thinking in LLMs I: Searching for the Right Intuition","date":"2025-02-10","arxiv_id":"2502.06773","n_code_links":4,"syntology":{"ran":11,"of":13,"n_ran_checked":11,"n_instrument":0,"unverified":2,"pointer_only":9,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["GuanghaoYe/Emergence-of-Thinking"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"fairness-aware-reinforcement-learning-via","title":"Fairness Aware Reinforcement Learning via Proximal Policy Optimization","date":"2025-02-06","arxiv_id":"2502.03953","n_code_links":0,"syntology":null},{"paper":null,"slug":"label-anything-an-interpretable-high-fidelity","title":"Label Anything: An Interpretable, High-Fidelity and Prompt-Free Annotator","date":"2025-02-05","arxiv_id":"2502.02972","n_code_links":0,"syntology":null},{"paper":null,"slug":"regret-optimized-portfolio-enhancement","title":"Regret-Optimized Portfolio Enhancement through Deep Reinforcement Learning and Future Looking Rewards","date":"2025-02-04","arxiv_id":"2502.02619","n_code_links":0,"syntology":null},{"paper":null,"slug":"mitigation-of-camouflaged-adversarial-attacks","title":"Mitigation of Camouflaged Adversarial Attacks in Autonomous Vehicles--A Case Study Using CARLA Simulator","date":"2025-02-03","arxiv_id":"2502.05208","n_code_links":0,"syntology":null},{"paper":"/paper/synthmanticlidar-a-synthetic-dataset-for","slug":"synthmanticlidar-a-synthetic-dataset-for","title":"SynthmanticLiDAR: A Synthetic Dataset for Semantic Segmentation on LiDAR Imaging","date":"2025-01-31","arxiv_id":"2501.19035","n_code_links":1,"syntology":null},{"paper":null,"slug":"the-energy-loss-phenomenon-in-rlhf-a-new","title":"The Energy Loss Phenomenon in RLHF: A New Perspective on Mitigating Reward Hacking","date":"2025-01-31","arxiv_id":"2501.19358","n_code_links":0,"syntology":null},{"paper":null,"slug":"hybrid-group-relative-policy-optimization-a","title":"Hybrid Group Relative Policy Optimization: A Multi-Sample Approach to Enhancing Policy Optimization","date":"2025-01-30","arxiv_id":"2502.01652","n_code_links":0,"syntology":null},{"paper":null,"slug":"ssf-pan-semantic-scene-flow-based-perception","title":"SSF-PAN: Semantic Scene Flow-Based Perception for Autonomous Navigation in Traffic Scenarios","date":"2025-01-28","arxiv_id":"2501.16754","n_code_links":0,"syntology":null},{"paper":"/paper/evorl-a-gpu-accelerated-framework-for","slug":"evorl-a-gpu-accelerated-framework-for","title":"EvoRL: A GPU-accelerated Framework for Evolutionary Reinforcement Learning","date":"2025-01-25","arxiv_id":"2501.15129","n_code_links":2,"syntology":null},{"paper":null,"slug":"adawm-adaptive-world-model-based-planning-for","title":"AdaWM: Adaptive World Model based Planning for Autonomous Driving","date":"2025-01-22","arxiv_id":"2501.13072","n_code_links":0,"syntology":null},{"paper":null,"slug":"heppo-hardware-efficient-proximal-policy","title":"HEPPO: Hardware-Efficient Proximal Policy Optimization -- A Universal Pipelined Architecture for Generalized Advantage Estimation","date":"2025-01-22","arxiv_id":"2501.12703","n_code_links":0,"syntology":null},{"paper":"/paper/to-measure-or-not-a-cost-sensitive-selective","slug":"to-measure-or-not-a-cost-sensitive-selective","title":"To Measure or Not: A Cost-Sensitive, Selective Measuring Environment for Agricultural Management Decisions with Reinforcement Learning","date":"2025-01-22","arxiv_id":"2501.12823","n_code_links":1,"syntology":null},{"paper":null,"slug":"explainable-lane-change-prediction-for-near","title":"Explainable Lane Change Prediction for Near-Crash Scenarios Using Knowledge Graph Embeddings and Retrieval Augmented Generation","date":"2025-01-20","arxiv_id":"2501.11560","n_code_links":0,"syntology":null},{"paper":null,"slug":"classical-and-deep-reinforcement-learning","title":"Classical and Deep Reinforcement Learning Inventory Control Policies for Pharmaceutical Supply Chains with Perishability and Non-Stationarity","date":"2025-01-18","arxiv_id":"2501.10895","n_code_links":0,"syntology":null},{"paper":"/paper/leapvad-a-leap-in-autonomous-driving-via","slug":"leapvad-a-leap-in-autonomous-driving-via","title":"LeapVAD: A Leap in Autonomous Driving via Cognitive Perception and Dual-Process Thinking","date":"2025-01-14","arxiv_id":"2501.08168","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":3,"n_instrument":1,"unverified":1,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"optimization-of-link-configuration-for","title":"Optimization of Link Configuration for Satellite Communication Using Reinforcement Learning","date":"2025-01-14","arxiv_id":"2501.08220","n_code_links":0,"syntology":null},{"paper":"/paper/a-hybrid-framework-for-reinsurance","slug":"a-hybrid-framework-for-reinsurance","title":"A Hybrid Framework for Reinsurance Optimization: Integrating Generative Models and Reinforcement Learning","date":"2025-01-11","arxiv_id":"2501.06404","n_code_links":1,"syntology":null},{"paper":null,"slug":"curla-curriculum-learning-based-deep","title":"CuRLA: Curriculum Learning Based Deep Reinforcement Learning for Autonomous Driving","date":"2025-01-09","arxiv_id":"2501.04982","n_code_links":0,"syntology":null},{"paper":null,"slug":"learningflow-automated-policy-learning","title":"LearningFlow: Automated Policy Learning Workflow for Urban Driving with Large Language Models","date":"2025-01-09","arxiv_id":"2501.05057","n_code_links":0,"syntology":null},{"paper":"/paper/reinforce-a-simple-and-efficient-approach-for","slug":"reinforce-a-simple-and-efficient-approach-for","title":"REINFORCE++: A Simple and Efficient Approach for Aligning Large Language Models","date":"2025-01-04","arxiv_id":"2501.03262","n_code_links":5,"syntology":null},{"paper":null,"slug":"drivegpt4-v2-harnessing-large-language-model","title":"DriveGPT4-V2: Harnessing Large Language Model Capabilities for Enhanced Closed-Loop Autonomous Driving","date":"2025-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/plug-and-play-ppo-an-adaptive-point-prompt","slug":"plug-and-play-ppo-an-adaptive-point-prompt","title":"Plug-and-Play PPO: An Adaptive Point Prompt Optimizer Making SAM Greater","date":"2025-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"vlms-guided-representation-distillation-for","title":"VLMs-Guided Representation Distillation for Efficient Vision-Based Reinforcement Learning","date":"2025-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"rfppo-motion-dynamic-rrt-based-fluid-field","title":"RFPPO: Motion Dynamic RRT based Fluid Field - PPO for Dynamic TF/TA Routing Planning","date":"2024-12-28","arxiv_id":"2412.20098","n_code_links":0,"syntology":null},{"paper":null,"slug":"graph-attention-based-casual-discovery-with","title":"Graph-attention-based Casual Discovery with Trust Region-navigated Clipping Policy Optimization","date":"2024-12-27","arxiv_id":"2412.19578","n_code_links":0,"syntology":null},{"paper":null,"slug":"preventive-energy-management-for-distribution","title":"Preventive Energy Management for Distribution Systems Under Uncertain Events: A Deep Reinforcement Learning Approach","date":"2024-12-26","arxiv_id":"2412.19382","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-learning-based-traffic-aware-base","title":"Deep Learning-Based Traffic-Aware Base Station Sleep Mode and Cell Zooming Strategy in RIS-Aided Multi-Cell Networks","date":"2024-12-25","arxiv_id":"2412.18983","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-multi-step-reasoning-abilities-of","title":"Improving Multi-Step Reasoning Abilities of Large Language Models with Direct Advantage Policy Optimization","date":"2024-12-24","arxiv_id":"2412.18279","n_code_links":0,"syntology":null},{"paper":null,"slug":"vlm-rl-a-unified-vision-language-models-and","title":"VLM-RL: A Unified Vision Language Models and Reinforcement Learning Framework for Safe Autonomous Driving","date":"2024-12-20","arxiv_id":"2412.15544","n_code_links":0,"syntology":null},{"paper":null,"slug":"active-reinforcement-learning-strategies-for","title":"Active Reinforcement Learning Strategies for Offline Policy Improvement","date":"2024-12-17","arxiv_id":"2412.13106","n_code_links":0,"syntology":null},{"paper":"/paper/efficient-policy-adaptation-with-contrastive-1","slug":"efficient-policy-adaptation-with-contrastive-1","title":"Efficient Policy Adaptation with Contrastive Prompt Ensemble for Embodied Agents","date":"2024-12-16","arxiv_id":"2412.11484","n_code_links":0,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":null}},{"paper":null,"slug":"automated-driving-with-evolution-capability-a","title":"Automated Driving with Evolution Capability: A Reinforcement Learning Method with Monotonic Performance Enhancement","date":"2024-12-14","arxiv_id":"2412.10822","n_code_links":0,"syntology":null},{"paper":null,"slug":"ei-drive-a-platform-for-cooperative","title":"EI-Drive: A Platform for Cooperative Perception with Realistic Communication Models","date":"2024-12-13","arxiv_id":"2412.09782","n_code_links":0,"syntology":null},{"paper":null,"slug":"trafficloc-localizing-traffic-surveillance","title":"TrafficLoc: Localizing Traffic Surveillance Cameras in 3D Scenes","date":"2024-12-13","arxiv_id":"2412.10308","n_code_links":0,"syntology":null},{"paper":"/paper/wisead-knowledge-augmented-end-to-end","slug":"wisead-knowledge-augmented-end-to-end","title":"WiseAD: Knowledge Augmented End-to-End Autonomous Driving with Vision-Language Model","date":"2024-12-13","arxiv_id":"2412.09951","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":3,"n_instrument":2,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["wyddmw/WiseAD"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/ecarla-scenes-a-synthetically-generated","slug":"ecarla-scenes-a-synthetically-generated","title":"eCARLA-scenes: A synthetically generated dataset for event-based optical flow prediction","date":"2024-12-12","arxiv_id":"2412.09209","n_code_links":2,"syntology":null},{"paper":"/paper/hidden-biases-of-end-to-end-driving-datasets","slug":"hidden-biases-of-end-to-end-driving-datasets","title":"Hidden Biases of End-to-End Driving Datasets","date":"2024-12-12","arxiv_id":"2412.09602","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["autonomousvision/carla_garage"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"bench2drive-r-turning-real-world-data-into","title":"Bench2Drive-R: Turning Real World Data into Reactive Closed-Loop Autonomous Driving Benchmark by Generative Model","date":"2024-12-11","arxiv_id":"2412.09647","n_code_links":0,"syntology":null},{"paper":"/paper/a-method-for-evaluating-hyperparameter","slug":"a-method-for-evaluating-hyperparameter","title":"A Method for Evaluating Hyperparameter Sensitivity in Reinforcement Learning","date":"2024-12-10","arxiv_id":"2412.07165","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["jadkins99/hyperparameter_sensitivity"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"a-scalable-decentralized-reinforcement","title":"A Scalable Decentralized Reinforcement Learning Framework for UAV Target Localization Using Recurrent PPO","date":"2024-12-09","arxiv_id":"2412.06231","n_code_links":0,"syntology":null},{"paper":null,"slug":"traffic-co-simulation-framework-empowered-by","title":"Traffic Co-Simulation Framework Empowered by Infrastructure Camera Sensing and Reinforcement Learning","date":"2024-12-05","arxiv_id":"2412.03925","n_code_links":0,"syntology":null},{"paper":null,"slug":"using-cooperative-co-evolutionary-search-to","title":"Using Cooperative Co-evolutionary Search to Generate Metamorphic Test Cases for Autonomous Driving Systems","date":"2024-12-05","arxiv_id":"2412.03843","n_code_links":0,"syntology":null},{"paper":"/paper/fedpaw-federated-learning-with-personalized","slug":"fedpaw-federated-learning-with-personalized","title":"FedPAW: Federated Learning with Personalized Aggregation Weights for Urban Vehicle Speed Prediction","date":"2024-12-02","arxiv_id":"2412.01281","n_code_links":1,"syntology":null},{"paper":null,"slug":"hprm-high-performance-robotic-middleware-for","title":"HPRM: High-Performance Robotic Middleware for Intelligent Autonomous Systems","date":"2024-12-02","arxiv_id":"2412.01799","n_code_links":0,"syntology":null},{"paper":null,"slug":"realistic-corner-case-generation-for","title":"Realistic Corner Case Generation for Autonomous Vehicles with Multimodal Large Language Model","date":"2024-11-29","arxiv_id":"2412.00243","n_code_links":0,"syntology":null},{"paper":null,"slug":"accelerating-proximal-policy-optimization","title":"Accelerating Proximal Policy Optimization Learning Using Task Prediction for Solving Environments with Delayed Rewards","date":"2024-11-26","arxiv_id":"2411.17861","n_code_links":0,"syntology":null},{"paper":null,"slug":"from-dashcam-videos-to-driving-simulations","title":"From Dashcam Videos to Driving Simulations: Stress Testing Automated Vehicles against Rare Events","date":"2024-11-25","arxiv_id":"2411.16027","n_code_links":0,"syntology":null},{"paper":null,"slug":"generating-out-of-distribution-scenarios","title":"Generating Out-Of-Distribution Scenarios Using Language Models","date":"2024-11-25","arxiv_id":"2411.16554","n_code_links":0,"syntology":null},{"paper":null,"slug":"imperceptible-adversarial-examples-in-the","title":"Imperceptible Adversarial Examples in the Physical World","date":"2024-11-25","arxiv_id":"2411.16622","n_code_links":0,"syntology":null},{"paper":null,"slug":"syndiff-ad-improving-semantic-segmentation","title":"SynDiff-AD: Improving Semantic Segmentation and End-to-End Autonomous Driving with Synthetic Data from Latent Diffusion Models","date":"2024-11-25","arxiv_id":"2411.16776","n_code_links":0,"syntology":null},{"paper":null,"slug":"reward-fine-tuning-two-step-diffusion-models","title":"Reward Fine-Tuning Two-Step Diffusion Models via Learning Differentiable Latent-Space Surrogate Reward","date":"2024-11-22","arxiv_id":"2411.15247","n_code_links":0,"syntology":null},{"paper":null,"slug":"provably-efficient-action-manipulation-attack","title":"Provably Efficient Action-Manipulation Attack Against Continuous Reinforcement Learning","date":"2024-11-20","arxiv_id":"2411.13116","n_code_links":0,"syntology":null},{"paper":"/paper/whales-a-multi-agent-scheduling-dataset-for","slug":"whales-a-multi-agent-scheduling-dataset-for","title":"WHALES: A Multi-agent Scheduling Dataset for Enhanced Cooperation in Autonomous Driving","date":"2024-11-20","arxiv_id":"2411.13340","n_code_links":1,"syntology":null},{"paper":null,"slug":"fast-convergence-of-softmax-policy-mirror","title":"Fast Convergence of Softmax Policy Mirror Ascent","date":"2024-11-18","arxiv_id":"2411.12042","n_code_links":0,"syntology":null},{"paper":null,"slug":"dynamics-of-resource-allocation-in-o-rans-an","title":"Dynamics of Resource Allocation in O-RANs: An In-depth Exploration of On-Policy and Off-Policy Deep Reinforcement Learning for Real-Time Applications","date":"2024-11-17","arxiv_id":"2412.01839","n_code_links":0,"syntology":null},{"paper":null,"slug":"imagine-2-drive-high-fidelity-world-modeling","title":"Imagine-2-Drive: Leveraging High-Fidelity World Models via Multi-Modal Diffusion Policies","date":"2024-11-15","arxiv_id":"2411.10171","n_code_links":0,"syntology":null},{"paper":null,"slug":"edge-caching-optimization-with-ppo-and","title":"Edge Caching Optimization with PPO and Transfer Learning for Dynamic Environments","date":"2024-11-14","arxiv_id":"2411.09812","n_code_links":0,"syntology":null},{"paper":null,"slug":"rationality-based-innate-values-driven","title":"Innate-Values-driven Reinforcement Learning based Cognitive Modeling","date":"2024-11-14","arxiv_id":"2411.09160","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-multi-agent-reinforcement-learning","title":"Exploring Multi-Agent Reinforcement Learning for Unrelated Parallel Machine Scheduling","date":"2024-11-12","arxiv_id":"2411.07634","n_code_links":0,"syntology":null},{"paper":null,"slug":"research-on-reinforcement-learning-based","title":"Research on reinforcement learning based warehouse robot navigation algorithm in complex warehouse layout","date":"2024-11-09","arxiv_id":"2411.06128","n_code_links":0,"syntology":null},{"paper":null,"slug":"querying-perception-streams-with-spatial","title":"Querying Perception Streams with Spatial Regular Expressions","date":"2024-11-08","arxiv_id":"2411.05946","n_code_links":0,"syntology":null},{"paper":"/paper/think-smart-act-smarl-analyzing-probabilistic","slug":"think-smart-act-smarl-analyzing-probabilistic","title":"Think Smart, Act SMARL! Analyzing Probabilistic Logic Shields for Multi-Agent Reinforcement Learning","date":"2024-11-07","arxiv_id":"2411.04867","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-comparative-study-of-deep-reinforcement-2","title":"A Comparative Study of Deep Reinforcement Learning for Crop Production Management","date":"2024-11-06","arxiv_id":"2411.04106","n_code_links":0,"syntology":null},{"paper":null,"slug":"federated-data-driven-kalman-filtering-for","title":"Federated Data-Driven Kalman Filtering for State Estimation","date":"2024-11-06","arxiv_id":"2411.05847","n_code_links":0,"syntology":null},{"paper":null,"slug":"salsa-soup-based-alignment-learning-for","title":"SALSA: Soup-based Alignment Learning for Stronger Adaptation in RLHF","date":"2024-11-04","arxiv_id":"2411.01798","n_code_links":0,"syntology":null},{"paper":null,"slug":"beyond-the-boundaries-of-proximal-policy","title":"Beyond the Boundaries of Proximal Policy Optimization","date":"2024-11-01","arxiv_id":"2411.00666","n_code_links":0,"syntology":null},{"paper":"/paper/cale-continuous-arcade-learning-environment","slug":"cale-continuous-arcade-learning-environment","title":"CALE: Continuous Arcade Learning Environment","date":"2024-10-31","arxiv_id":"2410.23810","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["farama-foundation/arcade-learning-environment"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"optical-lens-attack-on-monocular-depth","title":"Optical Lens Attack on Monocular Depth Estimation for Autonomous Driving","date":"2024-10-31","arxiv_id":"2411.00192","n_code_links":0,"syntology":null},{"paper":"/paper/mdcure-a-scalable-pipeline-for-multi-document","slug":"mdcure-a-scalable-pipeline-for-multi-document","title":"MDCure: A Scalable Pipeline for Multi-Document Instruction-Following","date":"2024-10-30","arxiv_id":"2410.23463","n_code_links":1,"syntology":null},{"paper":null,"slug":"self-driving-car-racing-application-of-deep","title":"Self-Driving Car Racing: Application of Deep Reinforcement Learning","date":"2024-10-30","arxiv_id":"2410.22766","n_code_links":0,"syntology":null},{"paper":null,"slug":"pre-trained-vision-models-as-perception","title":"Pre-Trained Vision Models as Perception Backbones for Safety Filters in Autonomous Driving","date":"2024-10-29","arxiv_id":"2410.22585","n_code_links":0,"syntology":null},{"paper":null,"slug":"ss3dm-benchmarking-street-view-surface","title":"SS3DM: Benchmarking Street-View Surface Reconstruction with a Synthetic 3D Mesh Dataset","date":"2024-10-29","arxiv_id":"2410.21739","n_code_links":0,"syntology":null},{"paper":null,"slug":"capacity-aware-planning-and-scheduling-in","title":"Capacity-Aware Planning and Scheduling in Budget-Constrained Monotonic MDPs: A Meta-RL Approach","date":"2024-10-28","arxiv_id":"2410.21249","n_code_links":0,"syntology":null},{"paper":null,"slug":"getting-by-goal-misgeneralization-with-a","title":"Getting By Goal Misgeneralization With a Little Help From a Mentor","date":"2024-10-28","arxiv_id":"2410.21052","n_code_links":0,"syntology":null},{"paper":"/paper/fast-best-of-n-decoding-via-speculative","slug":"fast-best-of-n-decoding-via-speculative","title":"Fast Best-of-N Decoding via Speculative Rejection","date":"2024-10-26","arxiv_id":"2410.20290","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":1,"n_instrument":3,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["Zanette-Labs/SpeculativeRejection"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"who-is-responsible-explaining-safety","title":"Who is Responsible? Explaining Safety Violations in Multi-Agent Cyber-Physical Systems","date":"2024-10-26","arxiv_id":"2410.20288","n_code_links":0,"syntology":null},{"paper":null,"slug":"aligning-codellms-with-direct-preference","title":"Aligning CodeLLMs with Direct Preference Optimization","date":"2024-10-24","arxiv_id":"2410.18585","n_code_links":0,"syntology":null},{"paper":"/paper/carla2real-a-tool-for-reducing-the-sim2real","slug":"carla2real-a-tool-for-reducing-the-sim2real","title":"CARLA2Real: a tool for reducing the sim2real gap in CARLA simulator","date":"2024-10-23","arxiv_id":"2410.18238","n_code_links":1,"syntology":null},{"paper":"/paper/exploring-rl-based-llm-training-for-formal","slug":"exploring-rl-based-llm-training-for-formal","title":"Exploring RL-based LLM Training for Formal Language Tasks with Programmed Rewards","date":"2024-10-22","arxiv_id":"2410.17126","n_code_links":1,"syntology":null},{"paper":null,"slug":"hierarchical-multi-agent-reinforcement-2","title":"Hierarchical Multi-agent Reinforcement Learning for Cyber Network Defense","date":"2024-10-22","arxiv_id":"2410.17351","n_code_links":0,"syntology":null},{"paper":null,"slug":"how-to-build-a-pre-trained-multimodal-model","title":"How to Build a Pre-trained Multimodal model for Simultaneously Chatting and Decision-making?","date":"2024-10-21","arxiv_id":"2410.15885","n_code_links":0,"syntology":null}],"record_sha256":"e86928300b983150644884660da5989a23439fb0aa0609133581d32fce6b359b","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}