{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/ppo/papers/5","list_of":"/method/ppo","method":"PPO","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":5,"pages_in_order":10,"rows_per_page":100,"rows":[401,500],"of":949,"counts":{"archive_papers_tagged":949,"with_a_code_link":397,"where_syntology_ran_a_sample":139,"not_listed_spam_title":0,"listed":949,"listed_where_code_ran":139,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":114,"every_run_a_failure_of_syntologys_instrument":25,"listed_with_a_run_with_no_instrument_failure":114,"listed_every_run_a_failure_of_syntologys_instrument":25,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/ppo","prev":"/method/ppo/papers/4","next":"/method/ppo/papers/6","papers":[{"paper":null,"slug":"safety-aware-autonomous-path-planning-using","title":"Safety Aware Autonomous Path Planning Using Model Predictive Reinforcement Learning for Inland Waterways","date":"2023-11-16","arxiv_id":"2311.09878","n_code_links":0,"syntology":null},{"paper":"/paper/clipped-objective-policy-gradients-for","slug":"clipped-objective-policy-gradients-for","title":"Clipped-Objective Policy Gradients for Pessimistic Policy Optimization","date":"2023-11-10","arxiv_id":"2311.05846","n_code_links":1,"syntology":null},{"paper":"/paper/black-box-prompt-optimization-aligning-large","slug":"black-box-prompt-optimization-aligning-large","title":"Black-Box Prompt Optimization: Aligning Large Language Models without Model Training","date":"2023-11-07","arxiv_id":"2311.04155","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":5,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["thu-coai/bpo"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"epidemic-decision-making-system-based","title":"Epidemic Decision-making System Based Federated Reinforcement Learning","date":"2023-11-03","arxiv_id":"2311.01749","n_code_links":0,"syntology":null},{"paper":null,"slug":"dropout-strategy-in-reinforcement-learning","title":"Dropout Strategy in Reinforcement Learning: Limiting the Surrogate Objective Variance in Policy Optimization Methods","date":"2023-10-31","arxiv_id":"2310.20380","n_code_links":0,"syntology":null},{"paper":null,"slug":"world-model-based-sim2real-transfer-for","title":"Bird's Eye View Based Pretrained World model for Visual Navigation","date":"2023-10-28","arxiv_id":"2310.18847","n_code_links":0,"syntology":null},{"paper":"/paper/reward-scale-robustness-for-proximal-policy","slug":"reward-scale-robustness-for-proximal-policy","title":"Reward Scale Robustness for Proximal Policy Optimization via DreamerV3 Tricks","date":"2023-10-26","arxiv_id":"2310.17805","n_code_links":0,"syntology":{"ran":4,"of":4,"n_ran_checked":3,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"cyclealign-iterative-distillation-from-black","title":"CycleAlign: Iterative Distillation from Black-box LLM to White-box Models for Better Human Alignment","date":"2023-10-25","arxiv_id":"2310.16271","n_code_links":0,"syntology":null},{"paper":"/paper/superhf-supervised-iterative-learning-from","slug":"superhf-supervised-iterative-learning-from","title":"SuperHF: Supervised Iterative Learning from Human Feedback","date":"2023-10-25","arxiv_id":"2310.16763","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["openfeedback/superhf"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/safe-navigation-training-autonomous-vehicles","slug":"safe-navigation-training-autonomous-vehicles","title":"Safe Navigation: Training Autonomous Vehicles using Deep Reinforcement Learning in CARLA","date":"2023-10-23","arxiv_id":"2311.10735","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["tejas-deo/safe-navigation-training-autonomous-vehicles-using-deep-reinforcement-learning-in-carla"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/automatic-unit-test-data-generation-and-actor","slug":"automatic-unit-test-data-generation-and-actor","title":"Automatic Unit Test Data Generation and Actor-Critic Reinforcement Learning for Code Synthesis","date":"2023-10-20","arxiv_id":"2310.13669","n_code_links":2,"syntology":null},{"paper":"/paper/letfuser-light-weight-end-to-end-transformer","slug":"letfuser-light-weight-end-to-end-transformer","title":"LeTFuser: Light-weight End-to-end Transformer-Based Sensor Fusion for Autonomous Driving with Multi-Task Learning","date":"2023-10-19","arxiv_id":"2310.13135","n_code_links":1,"syntology":null},{"paper":"/paper/remax-a-simple-effective-and-efficient-method","slug":"remax-a-simple-effective-and-efficient-method","title":"ReMax: A Simple, Effective, and Efficient Reinforcement Learning Method for Aligning Large Language Models","date":"2023-10-16","arxiv_id":"2310.10505","n_code_links":3,"syntology":{"ran":11,"of":13,"n_ran_checked":10,"n_instrument":1,"unverified":2,"pointer_only":13,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["liziniu/ReMax"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"optimizing-the-placement-of-roadside-lidars-1","title":"Optimizing the Placement of Roadside LiDARs for Autonomous Driving","date":"2023-10-11","arxiv_id":"2310.07247","n_code_links":0,"syntology":null},{"paper":"/paper/dsac-t-distributional-soft-actor-critic-with","slug":"dsac-t-distributional-soft-actor-critic-with","title":"Distributional Soft Actor-Critic with Three Refinements","date":"2023-10-09","arxiv_id":"2310.05858","n_code_links":2,"syntology":{"ran":5,"of":5,"n_ran_checked":4,"n_instrument":1,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["jingliang-duan/dsac-t","jingliang-duan/dsac-v2"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/reinforcement-learning-in-the-era-of-llms","slug":"reinforcement-learning-in-the-era-of-llms","title":"Reinforcement Learning in the Era of LLMs: What is Essential? What is needed? An RL Perspective on RLHF, Prompting, and Beyond","date":"2023-10-09","arxiv_id":"2310.06147","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/dialcot-meets-ppo-decomposing-and-exploring","slug":"dialcot-meets-ppo-decomposing-and-exploring","title":"DialCoT Meets PPO: Decomposing and Exploring Reasoning Paths in Smaller Language Models","date":"2023-10-08","arxiv_id":"2310.05074","n_code_links":1,"syntology":null},{"paper":null,"slug":"fp3o-enabling-proximal-policy-optimization-in","title":"FP3O: Enabling Proximal Policy Optimization in Multi-Agent Cooperation with Parameter-Sharing Versatility","date":"2023-10-08","arxiv_id":"2310.05053","n_code_links":0,"syntology":null},{"paper":null,"slug":"proximal-policy-optimization-based-1","title":"Proximal Policy Optimization-Based Reinforcement Learning Approach for DC-DC Boost Converter Control: A Comparative Evaluation Against Traditional Control Techniques","date":"2023-10-04","arxiv_id":"2310.02945","n_code_links":0,"syntology":null},{"paper":"/paper/reward-model-ensembles-help-mitigate","slug":"reward-model-ensembles-help-mitigate","title":"Reward Model Ensembles Help Mitigate Overoptimization","date":"2023-10-04","arxiv_id":"2310.02743","n_code_links":2,"syntology":{"ran":5,"of":5,"n_ran_checked":0,"n_instrument":5,"unverified":0,"pointer_only":3,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tlc4418/llm_optimization"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"adapting-llm-agents-through-communication","title":"Adapting LLM Agents with Universal Feedback in Communication","date":"2023-10-01","arxiv_id":"2310.01444","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-for-autonomous-6","title":"Deep Reinforcement Learning for Autonomous Vehicle Intersection Navigation","date":"2023-09-30","arxiv_id":"2310.08595","n_code_links":0,"syntology":null},{"paper":null,"slug":"pairwise-proximal-policy-optimization","title":"Pairwise Proximal Policy Optimization: Harnessing Relative Feedback for LLM Alignment","date":"2023-09-30","arxiv_id":"2310.00212","n_code_links":0,"syntology":null},{"paper":"/paper/carla-adjusted-common-average-referencing-for","slug":"carla-adjusted-common-average-referencing-for","title":"CARLA: Adjusted common average referencing for cortico-cortical evoked potential data","date":"2023-09-29","arxiv_id":"2310.00185","n_code_links":1,"syntology":null},{"paper":"/paper/cleanba-a-reproducible-and-efficient","slug":"cleanba-a-reproducible-and-efficient","title":"Cleanba: A Reproducible and Efficient Distributed Reinforcement Learning Platform","date":"2023-09-29","arxiv_id":"2310.00036","n_code_links":1,"syntology":{"ran":4,"of":8,"n_ran_checked":4,"n_instrument":0,"unverified":4,"pointer_only":8,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["vwxyzjn/cleanba"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/autonomous-driving-using-spiking-neural","slug":"autonomous-driving-using-spiking-neural","title":"Autonomous Driving using Spiking Neural Networks on Dynamic Vision Sensor Data: A Case Study of Traffic Light Change Detection","date":"2023-09-27","arxiv_id":"2311.09225","n_code_links":1,"syntology":null},{"paper":null,"slug":"making-ppo-even-better-value-guided-monte","title":"Don't throw away your value model! Generating more preferable text with Value-Guided Monte-Carlo Tree Search decoding","date":"2023-09-26","arxiv_id":"2309.15028","n_code_links":0,"syntology":null},{"paper":"/paper/enhancing-data-efficiency-in-reinforcement","slug":"enhancing-data-efficiency-in-reinforcement","title":"Enhancing data efficiency in reinforcement learning: a novel imagination mechanism based on mesh information propagation","date":"2023-09-25","arxiv_id":"2309.14243","n_code_links":2,"syntology":null},{"paper":null,"slug":"boosting-offline-reinforcement-learning-for","title":"Boosting Offline Reinforcement Learning for Autonomous Driving with Hierarchical Latent Skills","date":"2023-09-24","arxiv_id":"2309.13614","n_code_links":0,"syntology":null},{"paper":"/paper/hierarchical-adaptive-value-estimation-for","slug":"hierarchical-adaptive-value-estimation-for","title":"Hierarchical Adaptive Value Estimation for Multi-modal Visual Reinforcement Learning","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/learning-better-with-less-effective","slug":"learning-better-with-less-effective","title":"Learning Better with Less: Effective Augmentation for Sample-Efficient Visual Reinforcement Learning","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-to-drive-anywhere","title":"Learning to Drive Anywhere","date":"2023-09-21","arxiv_id":"2309.12295","n_code_links":0,"syntology":null},{"paper":"/paper/rrhf-rank-responses-to-align-language-models-1","slug":"rrhf-rank-responses-to-align-language-models-1","title":"RRHF: Rank Responses to Align Language Models with Human Feedback","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"ai-driven-patient-monitoring-with-multi-agent","title":"Adaptive Multi-Agent Deep Reinforcement Learning for Timely Healthcare Interventions","date":"2023-09-20","arxiv_id":"2309.10980","n_code_links":0,"syntology":null},{"paper":null,"slug":"privileged-to-predicted-towards-sensorimotor","title":"Privileged to Predicted: Towards Sensorimotor Reinforcement Learning for Urban Driving","date":"2023-09-18","arxiv_id":"2309.09756","n_code_links":0,"syntology":null},{"paper":null,"slug":"stabilizing-rlhf-through-advantage-model-and","title":"Stabilizing RLHF through Advantage Model and Selective Rehearsal","date":"2023-09-18","arxiv_id":"2309.10202","n_code_links":0,"syntology":null},{"paper":"/paper/exploring-the-impact-of-low-rank-adaptation","slug":"exploring-the-impact-of-low-rank-adaptation","title":"Exploring the impact of low-rank adaptation on the performance, efficiency, and regularization of RLHF","date":"2023-09-16","arxiv_id":"2309.09055","n_code_links":1,"syntology":{"ran":12,"of":13,"n_ran_checked":12,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["simengsun/alpaca_farm_lora"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"what-matters-to-enhance-traffic-rule","title":"What Matters to Enhance Traffic Rule Compliance of Imitation Learning for End-to-End Autonomous Driving","date":"2023-09-14","arxiv_id":"2309.07808","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-enabled-joint","title":"Deep Reinforcement Learning Enabled Joint Deployment and Beamforming in STAR-RIS Assisted Networks","date":"2023-09-07","arxiv_id":"2309.03520","n_code_links":0,"syntology":null},{"paper":null,"slug":"skope3d-a-synthetic-dataset-for-vehicle","title":"SKoPe3D: A Synthetic Dataset for Vehicle Keypoint Perception in 3D from Traffic Monitoring Cameras","date":"2023-09-04","arxiv_id":"2309.01324","n_code_links":0,"syntology":null},{"paper":null,"slug":"efficient-rlhf-reducing-the-memory-usage-of","title":"Efficient RLHF: Reducing the Memory Usage of PPO","date":"2023-09-01","arxiv_id":"2309.00754","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-reinforcement-learning-based-construction","title":"A reinforcement learning based construction material supply strategy using robotic crane and computer vision for building reconstruction after an earthquake","date":"2023-08-30","arxiv_id":"2308.16280","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-offline-evaluation-of-3d-object-detection","title":"On Offline Evaluation of 3D Object Detection for Autonomous Driving","date":"2023-08-24","arxiv_id":"2308.12779","n_code_links":0,"syntology":null},{"paper":"/paper/aligning-language-models-with-offline","slug":"aligning-language-models-with-offline","title":"Aligning Language Models with Offline Learning from Human Feedback","date":"2023-08-23","arxiv_id":"2308.12050","n_code_links":2,"syntology":null},{"paper":"/paper/pedestrian-environment-model-for-automated","slug":"pedestrian-environment-model-for-automated","title":"Pedestrian Environment Model for Automated Driving","date":"2023-08-17","arxiv_id":"2308.09080","n_code_links":1,"syntology":null},{"paper":null,"slug":"robust-autonomous-vehicle-pursuit-without","title":"Robust Autonomous Vehicle Pursuit without Expert Steering Labels","date":"2023-08-16","arxiv_id":"2308.08380","n_code_links":0,"syntology":null},{"paper":"/paper/a-deep-recurrent-reinforcement-learning","slug":"a-deep-recurrent-reinforcement-learning","title":"A Deep Recurrent-Reinforcement Learning Method for Intelligent AutoScaling of Serverless Functions","date":"2023-08-11","arxiv_id":"2308.05937","n_code_links":1,"syntology":null},{"paper":null,"slug":"proximal-policy-optimization-actual-combat","title":"Proximal Policy Optimization Actual Combat: Manipulating Output Tokenizer Length","date":"2023-08-10","arxiv_id":"2308.05585","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-the-physical-world-adversarial","title":"Exploring the Physical World Adversarial Robustness of Vehicle Detection","date":"2023-08-07","arxiv_id":"2308.03476","n_code_links":0,"syntology":null},{"paper":"/paper/semantics-guided-transformer-based-sensor","slug":"semantics-guided-transformer-based-sensor","title":"Cognitive TransFuser: Semantics-guided Transformer-based Sensor Fusion for Improved Waypoint Prediction","date":"2023-08-04","arxiv_id":"2308.02126","n_code_links":1,"syntology":null},{"paper":null,"slug":"interpretable-end-to-end-driving-model-for","title":"Interpretable End-to-End Driving Model for Implicit Scene Understanding","date":"2023-08-02","arxiv_id":"2308.01180","n_code_links":0,"syntology":null},{"paper":"/paper/parallel-q-learning-scaling-off-policy","slug":"parallel-q-learning-scaling-off-policy","title":"Parallel $Q$-Learning: Scaling Off-policy Reinforcement Learning under Massively Parallel Simulation","date":"2023-07-24","arxiv_id":"2307.12983","n_code_links":0,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/trust-aware-safe-control-for-autonomous","slug":"trust-aware-safe-control-for-autonomous","title":"Trust-aware Safe Control for Autonomous Navigation: Estimation of System-to-human Trust for Trust-adaptive Control Barrier Functions","date":"2023-07-24","arxiv_id":"2307.12815","n_code_links":3,"syntology":null},{"paper":"/paper/llama-2-open-foundation-and-fine-tuned-chat","slug":"llama-2-open-foundation-and-fine-tuned-chat","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","date":"2023-07-18","arxiv_id":"2307.09288","n_code_links":19,"syntology":{"ran":33,"of":52,"n_ran_checked":22,"n_instrument":11,"unverified":19,"pointer_only":20,"phrase":"33 ran (of which 7 constructed an object rather than computing a result; 22 with no instrument failure: 1 honoured, 1 violated, 20 with no contract checked; 11 where Syntology's instrument failed) · 19 unverified","official":{"repos":["facebookresearch/llama"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/navigating-uncertainty-the-role-of-short-term","slug":"navigating-uncertainty-the-role-of-short-term","title":"Navigating Uncertainty: The Role of Short-Term Trajectory Prediction in Autonomous Vehicle Safety","date":"2023-07-11","arxiv_id":"2307.05288","n_code_links":1,"syntology":null},{"paper":"/paper/secrets-of-rlhf-in-large-language-models-part","slug":"secrets-of-rlhf-in-large-language-models-part","title":"Secrets of RLHF in Large Language Models Part I: PPO","date":"2023-07-11","arxiv_id":"2307.04964","n_code_links":1,"syntology":null},{"paper":"/paper/discovering-hierarchical-achievements-in-1","slug":"discovering-hierarchical-achievements-in-1","title":"Discovering Hierarchical Achievements in Reinforcement Learning via Contrastive Learning","date":"2023-07-07","arxiv_id":"2307.03486","n_code_links":1,"syntology":null},{"paper":"/paper/containergym-a-real-world-reinforcement","slug":"containergym-a-real-world-reinforcement","title":"ContainerGym: A Real-World Reinforcement Learning Benchmark for Resource Allocation","date":"2023-07-06","arxiv_id":"2307.02991","n_code_links":1,"syntology":null},{"paper":null,"slug":"navigation-of-micro-robot-swarms-for-targeted","title":"Navigation of micro-robot swarms for targeted delivery using reinforcement learning","date":"2023-06-30","arxiv_id":"2306.17598","n_code_links":0,"syntology":null},{"paper":"/paper/preference-ranking-optimization-for-human","slug":"preference-ranking-optimization-for-human","title":"Preference Ranking Optimization for Human Alignment","date":"2023-06-30","arxiv_id":"2306.17492","n_code_links":1,"syntology":null},{"paper":null,"slug":"action-and-trajectory-planning-for-urban","title":"Action and Trajectory Planning for Urban Autonomous Driving with Hierarchical Reinforcement Learning","date":"2023-06-28","arxiv_id":"2306.15968","n_code_links":0,"syntology":null},{"paper":"/paper/creating-valid-adversarial-examples-of","slug":"creating-valid-adversarial-examples-of","title":"Creating Valid Adversarial Examples of Malware","date":"2023-06-23","arxiv_id":"2306.13587","n_code_links":1,"syntology":null},{"paper":null,"slug":"autonomous-driving-with-deep-reinforcement","title":"Autonomous Driving with Deep Reinforcement Learning in CARLA Simulation","date":"2023-06-20","arxiv_id":"2306.11217","n_code_links":0,"syntology":null},{"paper":"/paper/learning-to-generate-better-than-your-llm","slug":"learning-to-generate-better-than-your-llm","title":"Learning to Generate Better Than Your LLM","date":"2023-06-20","arxiv_id":"2306.11816","n_code_links":1,"syntology":null},{"paper":null,"slug":"safe-efficient-comfort-and-energy-saving","title":"Safe, Efficient, Comfort, and Energy-saving Automated Driving through Roundabout Based on Deep Reinforcement Learning","date":"2023-06-20","arxiv_id":"2306.11465","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-study-on-quantifying-sim2real-image-gap-in","title":"A Study on Quantifying Sim2Real Image Gap in Autonomous Driving Simulations Using Lane Segmentation Attention Map Similarity","date":"2023-06-18","arxiv_id":"2306.10491","n_code_links":0,"syntology":null},{"paper":"/paper/coaching-a-teachable-student-1","slug":"coaching-a-teachable-student-1","title":"Coaching a Teachable Student","date":"2023-06-16","arxiv_id":"2306.10014","n_code_links":1,"syntology":{"ran":0,"of":4,"n_ran_checked":0,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"0 ran · 4 unverified","official":{"repos":["h2xlab/CaT"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":[]}}},{"paper":"/paper/hidden-biases-of-end-to-end-driving-models","slug":"hidden-biases-of-end-to-end-driving-models","title":"Hidden Biases of End-to-End Driving Models","date":"2023-06-13","arxiv_id":"2306.07957","n_code_links":1,"syntology":{"ran":13,"of":17,"n_ran_checked":10,"n_instrument":3,"unverified":4,"pointer_only":0,"phrase":"13 ran (of which 10 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","official":{"repos":["autonomousvision/carla_garage"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":10,"n_ran_no_instrument_failure":10,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/backproptools-a-fast-portable-deep","slug":"backproptools-a-fast-portable-deep","title":"RLtools: A Fast, Portable Deep Reinforcement Learning Library for Continuous Control","date":"2023-06-06","arxiv_id":"2306.03530","n_code_links":1,"syntology":null},{"paper":"/paper/fine-tuning-language-models-with-advantage","slug":"fine-tuning-language-models-with-advantage","title":"Fine-Tuning Language Models with Advantage-Induced Policy Alignment","date":"2023-06-04","arxiv_id":"2306.02231","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":5,"n_instrument":0,"unverified":2,"pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["microsoft/rlhf-apa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"deep-q-learning-versus-proximal-policy","title":"Deep Q-Learning versus Proximal Policy Optimization: Performance Comparison in a Material Sorting Task","date":"2023-06-02","arxiv_id":"2306.01451","n_code_links":0,"syntology":null},{"paper":"/paper/relu-to-the-rescue-improve-your-on-policy","slug":"relu-to-the-rescue-improve-your-on-policy","title":"ReLU to the Rescue: Improve Your On-Policy Actor-Critic with Positive Advantages","date":"2023-06-02","arxiv_id":"2306.01460","n_code_links":1,"syntology":null},{"paper":"/paper/normalization-enhances-generalization-in","slug":"normalization-enhances-generalization-in","title":"Normalization Enhances Generalization in Visual Reinforcement Learning","date":"2023-06-01","arxiv_id":"2306.00656","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":0,"n_instrument":2,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["lilucse/Normalization-Enhances-Generalization-in-Visual-Reinforcement-Learning"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/latent-exploration-for-reinforcement-learning","slug":"latent-exploration-for-reinforcement-learning","title":"Latent Exploration for Reinforcement Learning","date":"2023-05-31","arxiv_id":"2305.20065","n_code_links":1,"syntology":null},{"paper":null,"slug":"resilience-in-platoons-of-cooperative","title":"Resilience in Platoons of Cooperative Heterogeneous Vehicles: Self-organization Strategies and Provably-correct Design","date":"2023-05-27","arxiv_id":"2305.17443","n_code_links":0,"syntology":null},{"paper":"/paper/generating-synergistic-formulaic-alpha","slug":"generating-synergistic-formulaic-alpha","title":"Generating Synergistic Formulaic Alpha Collections via Reinforcement Learning","date":"2023-05-25","arxiv_id":"2306.12964","n_code_links":1,"syntology":null},{"paper":"/paper/realistically-distributing-object-placements","slug":"realistically-distributing-object-placements","title":"Realistically distributing object placements in synthetic training data improves the performance of vision-based object detection models","date":"2023-05-24","arxiv_id":"2305.14621","n_code_links":1,"syntology":null},{"paper":"/paper/alpacafarm-a-simulation-framework-for-methods-1","slug":"alpacafarm-a-simulation-framework-for-methods-1","title":"AlpacaFarm: A Simulation Framework for Methods that Learn from Human Feedback","date":"2023-05-22","arxiv_id":"2305.14387","n_code_links":2,"syntology":null},{"paper":null,"slug":"learning-pedestrian-actions-to-ensure-safe","title":"Learning Pedestrian Actions to Ensure Safe Autonomous Driving","date":"2023-05-22","arxiv_id":"2305.13051","n_code_links":0,"syntology":null},{"paper":"/paper/actor-critic-methods-using-physics-informed","slug":"actor-critic-methods-using-physics-informed","title":"Actor-Critic Methods using Physics-Informed Neural Networks: Control of a 1D PDE Model for Fluid-Cooled Battery Packs","date":"2023-05-18","arxiv_id":"2305.10952","n_code_links":1,"syntology":null},{"paper":"/paper/reasonnet-end-to-end-driving-with-temporal-1","slug":"reasonnet-end-to-end-driving-with-temporal-1","title":"ReasonNet: End-to-End Driving with Temporal and Global Reasoning","date":"2023-05-17","arxiv_id":"2305.10507","n_code_links":0,"syntology":null},{"paper":null,"slug":"slic-hf-sequence-likelihood-calibration-with","title":"SLiC-HF: Sequence Likelihood Calibration with Human Feedback","date":"2023-05-17","arxiv_id":"2305.10425","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-theoretical-analysis-of-optimistic-proximal","title":"A Theoretical Analysis of Optimistic Proximal Policy Optimization in Linear Markov Decision Processes","date":"2023-05-15","arxiv_id":"2305.08841","n_code_links":0,"syntology":null},{"paper":null,"slug":"dynamically-conservative-self-driving-planner","title":"Dynamically Conservative Self-Driving Planner for Long-Tail Cases","date":"2023-05-12","arxiv_id":"2305.07497","n_code_links":0,"syntology":null},{"paper":"/paper/think-twice-before-driving-towards-scalable","slug":"think-twice-before-driving-towards-scalable","title":"Think Twice before Driving: Towards Scalable Decoders for End-to-End Autonomous Driving","date":"2023-05-10","arxiv_id":"2305.06242","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["opendrivelab/thinktwice"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/reducing-the-cost-of-cycle-time-tuning-for","slug":"reducing-the-cost-of-cycle-time-tuning-for","title":"Reducing the Cost of Cycle-Time Tuning for Real-World Policy Optimization","date":"2023-05-09","arxiv_id":"2305.05760","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["homayoonfarrahi/cycle-time-study"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/local-optimization-achieves-global-optimality","slug":"local-optimization-achieves-global-optimality","title":"Local Optimization Achieves Global Optimality in Multi-Agent Reinforcement Learning","date":"2023-05-08","arxiv_id":"2305.04819","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["zhaoyl18/ratio_game"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/carla-bsp-a-simulated-dataset-with","slug":"carla-bsp-a-simulated-dataset-with","title":"CARLA-BSP: a simulated dataset with pedestrians","date":"2023-04-29","arxiv_id":"2305.00204","n_code_links":2,"syntology":null},{"paper":null,"slug":"adversarial-policy-optimization-in-deep","title":"Adversarial Policy Optimization in Deep Reinforcement Learning","date":"2023-04-27","arxiv_id":"2304.14533","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-agents-run-relay-race-with-strangers","title":"Can Agents Run Relay Race with Strangers? Generalization of RL to Out-of-Distribution Trajectories","date":"2023-04-26","arxiv_id":"2304.13424","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-end-to-end-vehicle-trajcetory-prediction","title":"An End-to-End Vehicle Trajcetory Prediction Framework","date":"2023-04-19","arxiv_id":"2304.09764","n_code_links":0,"syntology":null},{"paper":"/paper/bridging-rl-theory-and-practice-with-the-1","slug":"bridging-rl-theory-and-practice-with-the-1","title":"Bridging RL Theory and Practice with the Effective Horizon","date":"2023-04-19","arxiv_id":"2304.09853","n_code_links":1,"syntology":null},{"paper":null,"slug":"benchmarking-the-physical-world-adversarial","title":"Benchmarking the Physical-world Adversarial Robustness of Vehicle Detection","date":"2023-04-11","arxiv_id":"2304.05098","n_code_links":0,"syntology":null},{"paper":"/paper/rrhf-rank-responses-to-align-language-models","slug":"rrhf-rank-responses-to-align-language-models","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","date":"2023-04-11","arxiv_id":"2304.05302","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ganjinzero/rrhf"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"lane-lighting-aware-neural-fields-for","title":"LANe: Lighting-Aware Neural Fields for Compositional Scene Synthesis","date":"2023-04-06","arxiv_id":"2304.03280","n_code_links":0,"syntology":null},{"paper":"/paper/autorl-hyperparameter-landscapes","slug":"autorl-hyperparameter-landscapes","title":"AutoRL Hyperparameter Landscapes","date":"2023-04-05","arxiv_id":"2304.02396","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["automl/autorl-landscape"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"pac-based-formal-verification-for-out-of","title":"PAC-Based Formal Verification for Out-of-Distribution Data Detection","date":"2023-04-04","arxiv_id":"2304.01592","n_code_links":0,"syntology":null},{"paper":null,"slug":"understanding-reinforcement-learning","title":"Understanding Reinforcement Learning Algorithms: The Progress from Basic Q-learning to Proximal Policy Optimization","date":"2023-03-31","arxiv_id":"2304.00026","n_code_links":0,"syntology":null},{"paper":null,"slug":"specification-guided-data-aggregation-for","title":"Specification-Guided Data Aggregation for Semantically Aware Imitation Learning","date":"2023-03-29","arxiv_id":"2303.17010","n_code_links":0,"syntology":null},{"paper":"/paper/model-based-reinforcement-learning-with-3","slug":"model-based-reinforcement-learning-with-3","title":"Model-Based Reinforcement Learning with Isolated Imaginations","date":"2023-03-27","arxiv_id":"2303.14889","n_code_links":1,"syntology":null}],"record_sha256":"7142f5fafaeee02bf161bbf3bde30296743d306d21f2015357015d4cf3d41d54","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}