{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/ppo/papers/4","list_of":"/method/ppo","method":"PPO","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":4,"pages_in_order":10,"rows_per_page":100,"rows":[301,400],"of":949,"counts":{"archive_papers_tagged":949,"with_a_code_link":397,"where_syntology_ran_a_sample":139,"not_listed_spam_title":0,"listed":949,"listed_where_code_ran":139,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":114,"every_run_a_failure_of_syntologys_instrument":25,"listed_with_a_run_with_no_instrument_failure":114,"listed_every_run_a_failure_of_syntologys_instrument":25,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/ppo","prev":"/method/ppo/papers/3","next":"/method/ppo/papers/5","papers":[{"paper":"/paper/continuously-learning-adapting-and-improving","slug":"continuously-learning-adapting-and-improving","title":"Continuously Learning, Adapting, and Improving: A Dual-Process Approach to Autonomous Driving","date":"2024-05-24","arxiv_id":"2405.15324","n_code_links":1,"syntology":null},{"paper":null,"slug":"efficient-reinforcement-learning-via-large","title":"Extracting Heuristics from Large Language Models for Reward Shaping in Reinforcement Learning","date":"2024-05-24","arxiv_id":"2405.15194","n_code_links":0,"syntology":null},{"paper":"/paper/agile-a-novel-framework-of-llm-agents","slug":"agile-a-novel-framework-of-llm-agents","title":"AGILE: A Novel Reinforcement Learning Framework of LLM Agents","date":"2024-05-23","arxiv_id":"2405.14751","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":3,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["bytarnish/agile"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/chatscene-knowledge-enabled-safety-critical","slug":"chatscene-knowledge-enabled-safety-critical","title":"ChatScene: Knowledge-Enabled Safety-Critical Scenario Generation for Autonomous Vehicles","date":"2024-05-22","arxiv_id":"2405.14062","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["javyduck/ChatScene"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"maskfuser-masked-fusion-of-joint-multi-modal","title":"MaskFuser: Masked Fusion of Joint Multi-Modal Tokenization for End-to-End Autonomous Driving","date":"2024-05-13","arxiv_id":"2405.07573","n_code_links":0,"syntology":null},{"paper":"/paper/value-augmented-sampling-for-language-model","slug":"value-augmented-sampling-for-language-model","title":"Value Augmented Sampling for Language Model Alignment and Personalization","date":"2024-05-10","arxiv_id":"2405.06639","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["idanshen/Value-Augmented-Sampling"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/proximal-policy-optimization-with-adaptive-1","slug":"proximal-policy-optimization-with-adaptive-1","title":"Proximal Policy Optimization with Adaptive Exploration","date":"2024-05-07","arxiv_id":"2405.04664","n_code_links":1,"syntology":null},{"paper":null,"slug":"guidance-design-for-escape-flight-vehicle","title":"Guidance Design for Escape Flight Vehicle Using Evolution Strategy Enhanced Deep Reinforcement Learning","date":"2024-05-04","arxiv_id":"2405.03711","n_code_links":0,"syntology":null},{"paper":"/paper/d2po-discriminator-guided-dpo-with-response","slug":"d2po-discriminator-guided-dpo-with-response","title":"D2PO: Discriminator-Guided DPO with Response Evaluation Models","date":"2024-05-02","arxiv_id":"2405.01511","n_code_links":1,"syntology":{"ran":11,"of":13,"n_ran_checked":11,"n_instrument":0,"unverified":2,"pointer_only":13,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["PrasannS/d2po"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/no-representation-no-trust-connecting","slug":"no-representation-no-trust-connecting","title":"No Representation, No Trust: Connecting Representation, Collapse, and Trust Issues in PPO","date":"2024-05-01","arxiv_id":"2405.00662","n_code_links":1,"syntology":{"ran":6,"of":8,"n_ran_checked":5,"n_instrument":1,"unverified":2,"pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["claire-labo/no-representation-no-trust"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/guiding-attention-in-end-to-end-driving","slug":"guiding-attention-in-end-to-end-driving","title":"Guiding Attention in End-to-End Driving Models","date":"2024-04-30","arxiv_id":"2405.00242","n_code_links":1,"syntology":null},{"paper":"/paper/dpo-meets-ppo-reinforced-token-optimization","slug":"dpo-meets-ppo-reinforced-token-optimization","title":"DPO Meets PPO: Reinforced Token Optimization for RLHF","date":"2024-04-29","arxiv_id":"2404.18922","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":3,"n_instrument":1,"unverified":2,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["zkshan2002/rto"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/3d-extended-object-tracking-by-fusing","slug":"3d-extended-object-tracking-by-fusing","title":"3D Extended Object Tracking by Fusing Roadside Sparse Radar Point Clouds and Pixel Keypoints","date":"2024-04-27","arxiv_id":"2404.17903","n_code_links":2,"syntology":null},{"paper":"/paper/rebel-reinforcement-learning-via-regressing","slug":"rebel-reinforcement-learning-via-regressing","title":"REBEL: Reinforcement Learning via Regressing Relative Rewards","date":"2024-04-25","arxiv_id":"2404.16767","n_code_links":3,"syntology":{"ran":16,"of":20,"n_ran_checked":12,"n_instrument":4,"unverified":4,"pointer_only":6,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 0 violated, 11 with no contract checked; 4 where Syntology's instrument failed) · 4 unverified","official":{"repos":["Owen-Oertell/rlcm","zhaolingao/rebel"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"contextualfusion-context-based-multi-sensor","title":"ContextualFusion: Context-Based Multi-Sensor Fusion for 3D Object Detection in Adverse Operating Conditions","date":"2024-04-23","arxiv_id":"2404.14780","n_code_links":0,"syntology":null},{"paper":"/paper/is-dpo-superior-to-ppo-for-llm-alignment-a","slug":"is-dpo-superior-to-ppo-for-llm-alignment-a","title":"Is DPO Superior to PPO for LLM Alignment? A Comprehensive Study","date":"2024-04-16","arxiv_id":"2404.10719","n_code_links":1,"syntology":null},{"paper":"/paper/sevd-synthetic-event-based-vision-dataset-for","slug":"sevd-synthetic-event-based-vision-dataset-for","title":"SEVD: Synthetic Event-based Vision Dataset for Ego and Fixed Traffic Perception","date":"2024-04-12","arxiv_id":"2404.10540","n_code_links":1,"syntology":null},{"paper":null,"slug":"enhanced-cooperative-perception-for","title":"Enhanced Cooperative Perception for Autonomous Vehicles Using Imperfect Communication","date":"2024-04-10","arxiv_id":"2404.08013","n_code_links":0,"syntology":null},{"paper":null,"slug":"synergy-of-large-language-model-and-model","title":"Synergy of Large Language Model and Model Driven Engineering for Automated Development of Centralized Vehicular Systems","date":"2024-04-08","arxiv_id":"2404.05508","n_code_links":0,"syntology":null},{"paper":null,"slug":"prompting-multi-modal-tokens-to-enhance-end","title":"Prompting Multi-Modal Tokens to Enhance End-to-End Autonomous Driving Imitation Learning with LLMs","date":"2024-04-07","arxiv_id":"2404.04869","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-proximal-policy-optimization-based","title":"A proximal policy optimization based intelligent home solar management","date":"2024-04-05","arxiv_id":"2404.03888","n_code_links":0,"syntology":null},{"paper":"/paper/agl-net-aerial-ground-cross-modal-global","slug":"agl-net-aerial-ground-cross-modal-global","title":"AGL-NET: Aerial-Ground Cross-Modal Global Localization with Varying Scales","date":"2024-04-04","arxiv_id":"2404.03187","n_code_links":1,"syntology":null},{"paper":"/paper/addressing-loss-of-plasticity-and","slug":"addressing-loss-of-plasticity-and","title":"Addressing Loss of Plasticity and Catastrophic Forgetting in Continual Learning","date":"2024-03-31","arxiv_id":"2404.00781","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["mohmdelsayed/upgd"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/human-compatible-driving-partners-through","slug":"human-compatible-driving-partners-through","title":"Human-compatible driving partners through data-regularized self-play reinforcement learning","date":"2024-03-28","arxiv_id":"2403.19648","n_code_links":1,"syntology":null},{"paper":"/paper/scenario-based-curriculum-generation-for","slug":"scenario-based-curriculum-generation-for","title":"Scenario-Based Curriculum Generation for Multi-Agent Autonomous Driving","date":"2024-03-26","arxiv_id":"2403.17805","n_code_links":1,"syntology":null},{"paper":null,"slug":"drivecot-integrating-chain-of-thought","title":"DriveCoT: Integrating Chain-of-Thought Reasoning with End-to-End Driving","date":"2024-03-25","arxiv_id":"2403.16996","n_code_links":0,"syntology":null},{"paper":"/paper/policy-mirror-descent-with-lookahead","slug":"policy-mirror-descent-with-lookahead","title":"Policy Mirror Descent with Lookahead","date":"2024-03-21","arxiv_id":"2403.14156","n_code_links":1,"syntology":null},{"paper":"/paper/equivariant-ensembles-and-regularization-for","slug":"equivariant-ensembles-and-regularization-for","title":"Equivariant Ensembles and Regularization for Reinforcement Learning in Map-based Path Planning","date":"2024-03-19","arxiv_id":"2403.12856","n_code_links":1,"syntology":null},{"paper":"/paper/jaxued-a-simple-and-useable-ued-library-in","slug":"jaxued-a-simple-and-useable-ued-library-in","title":"JaxUED: A simple and useable UED library in Jax","date":"2024-03-19","arxiv_id":"2403.13091","n_code_links":1,"syntology":null},{"paper":null,"slug":"m2da-multi-modal-fusion-transformer","title":"M2DA: Multi-Modal Fusion Transformer Incorporating Driver Attention for Autonomous Driving","date":"2024-03-19","arxiv_id":"2403.12552","n_code_links":0,"syntology":null},{"paper":"/paper/driving-style-alignment-for-llm-powered","slug":"driving-style-alignment-for-llm-powered","title":"Driving Style Alignment for LLM-powered Driver Agent","date":"2024-03-17","arxiv_id":"2403.11368","n_code_links":2,"syntology":null},{"paper":null,"slug":"are-you-a-robot-detecting-autonomous-vehicles","title":"Are you a robot? Detecting Autonomous Vehicles from Behavior Analysis","date":"2024-03-14","arxiv_id":"2403.09571","n_code_links":0,"syntology":null},{"paper":null,"slug":"right-place-right-time-towards-objectnav-for","title":"Right Place, Right Time! Dynamizing Topological Graphs for Embodied Navigation","date":"2024-03-14","arxiv_id":"2403.09905","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-reinforcement-learning-from-human-1","title":"Improving Reinforcement Learning from Human Feedback Using Contrastive Rewards","date":"2024-03-12","arxiv_id":"2403.07708","n_code_links":0,"syntology":null},{"paper":null,"slug":"tractable-joint-prediction-and-planning-over","title":"Tractable Joint Prediction and Planning over Discrete Behavior Modes for Urban Driving","date":"2024-03-12","arxiv_id":"2403.07232","n_code_links":0,"syntology":null},{"paper":null,"slug":"risk-sensitive-rl-with-optimized-certainty","title":"Risk-Sensitive RL with Optimized Certainty Equivalents via Reduction to Standard RL","date":"2024-03-10","arxiv_id":"2403.06323","n_code_links":0,"syntology":null},{"paper":null,"slug":"teaching-large-language-models-to-reason-with","title":"Teaching Large Language Models to Reason with Reinforcement Learning","date":"2024-03-07","arxiv_id":"2403.04642","n_code_links":0,"syntology":null},{"paper":null,"slug":"commit-certifying-robustness-of-multi-sensor","title":"COMMIT: Certifying Robustness of Multi-Sensor Fusion Systems against Semantic Attacks","date":"2024-03-04","arxiv_id":"2403.02329","n_code_links":0,"syntology":null},{"paper":"/paper/snapshot-reinforcement-learning-leveraging","slug":"snapshot-reinforcement-learning-leveraging","title":"Snapshot Reinforcement Learning: Leveraging Prior Trajectories for Efficiency","date":"2024-03-01","arxiv_id":"2403.00673","n_code_links":1,"syntology":null},{"paper":"/paper/craftax-a-lightning-fast-benchmark-for-open","slug":"craftax-a-lightning-fast-benchmark-for-open","title":"Craftax: A Lightning-Fast Benchmark for Open-Ended Reinforcement Learning","date":"2024-02-26","arxiv_id":"2402.16801","n_code_links":1,"syntology":{"ran":2,"of":16,"n_ran_checked":2,"n_instrument":0,"unverified":14,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 14 unverified","official":{"repos":["michaeltmatthews/craftax"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":14,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"back-to-basics-revisiting-reinforce-style","title":"Back to Basics: Revisiting REINFORCE Style Optimization for Learning from Human Feedback in LLMs","date":"2024-02-22","arxiv_id":"2402.14740","n_code_links":0,"syntology":null},{"paper":null,"slug":"distributed-radiance-fields-for-edge-video","title":"Distributed Radiance Fields for Edge Video Compression and Metaverse Integration in Autonomous Driving","date":"2024-02-22","arxiv_id":"2402.14642","n_code_links":0,"syntology":null},{"paper":"/paper/hybrid-reasoning-based-on-large-language","slug":"hybrid-reasoning-based-on-large-language","title":"Hybrid Reasoning Based on Large Language Models for Autonomous Car Driving","date":"2024-02-21","arxiv_id":"2402.13602","n_code_links":1,"syntology":null},{"paper":"/paper/vadv2-end-to-end-vectorized-autonomous","slug":"vadv2-end-to-end-vectorized-autonomous","title":"VADv2: End-to-End Vectorized Autonomous Driving via Probabilistic Planning","date":"2024-02-20","arxiv_id":"2402.13243","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["hustvl/vad"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["community","official"]}}},{"paper":null,"slug":"surpassing-legacy-approaches-and-human","title":"Surpassing legacy approaches to PWR core reload optimization with single-objective Reinforcement learning","date":"2024-02-16","arxiv_id":"2402.11040","n_code_links":0,"syntology":null},{"paper":null,"slug":"rs-dpo-a-hybrid-rejection-sampling-and-direct","title":"RS-DPO: A Hybrid Rejection Sampling and Direct Preference Optimization Method for Alignment of Large Language Models","date":"2024-02-15","arxiv_id":"2402.10038","n_code_links":0,"syntology":null},{"paper":"/paper/reducing-texture-bias-of-deep-neural-networks","slug":"reducing-texture-bias-of-deep-neural-networks","title":"Reducing Texture Bias of Deep Neural Networks via Edge Enhancing Diffusion","date":"2024-02-14","arxiv_id":"2402.09530","n_code_links":1,"syntology":null},{"paper":null,"slug":"semtra-a-semantic-skill-translator-for-cross","title":"SemTra: A Semantic Skill Translator for Cross-Domain Zero-Shot Policy Adaptation","date":"2024-02-12","arxiv_id":"2402.07418","n_code_links":0,"syntology":null},{"paper":"/paper/solving-deep-reinforcement-learning","slug":"solving-deep-reinforcement-learning","title":"Solving Deep Reinforcement Learning Tasks with Evolution Strategies and Linear Policy Networks","date":"2024-02-10","arxiv_id":"2402.06912","n_code_links":1,"syntology":null},{"paper":"/paper/entropy-regularized-token-level-policy","slug":"entropy-regularized-token-level-policy","title":"Entropy-Regularized Token-Level Policy Optimization for Language Agent Reinforcement","date":"2024-02-09","arxiv_id":"2402.06700","n_code_links":1,"syntology":{"ran":6,"of":9,"n_ran_checked":3,"n_instrument":3,"unverified":3,"pointer_only":9,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","official":{"repos":["morning9393/etpo"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"edge-caching-based-on-deep-reinforcement","title":"Attention-Enhanced Prioritized Proximal Policy Optimization for Adaptive Edge Caching","date":"2024-02-08","arxiv_id":"2402.14576","n_code_links":0,"syntology":null},{"paper":null,"slug":"convergence-for-natural-policy-gradient-on","title":"Convergence for Natural Policy Gradient on Infinite-State Queueing MDPs","date":"2024-02-07","arxiv_id":"2402.05274","n_code_links":0,"syntology":null},{"paper":null,"slug":"tuning-the-feedback-controller-gains-is-a","title":"Tuning the feedback controller gains is a simple way to improve autonomous driving performance","date":"2024-02-07","arxiv_id":"2402.05064","n_code_links":0,"syntology":null},{"paper":"/paper/compound-returns-reduce-variance-in","slug":"compound-returns-reduce-variance-in","title":"Averaging $n$-step Returns Reduces Variance in Reinforcement Learning","date":"2024-02-06","arxiv_id":"2402.03903","n_code_links":0,"syntology":{"ran":5,"of":8,"n_ran_checked":4,"n_instrument":1,"unverified":3,"pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":null}},{"paper":"/paper/learning-to-generate-explainable-stock","slug":"learning-to-generate-explainable-stock","title":"Learning to Generate Explainable Stock Predictions using Self-Reflective Large Language Models","date":"2024-02-06","arxiv_id":"2402.03659","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["koa-fin/sep"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/oasim-an-open-and-adaptive-simulator-based-on","slug":"oasim-an-open-and-adaptive-simulator-based-on","title":"OASim: an Open and Adaptive Simulator based on Neural Rendering for Autonomous Driving","date":"2024-02-06","arxiv_id":"2402.03830","n_code_links":1,"syntology":null},{"paper":"/paper/deepseekmath-pushing-the-limits-of","slug":"deepseekmath-pushing-the-limits-of","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","date":"2024-02-05","arxiv_id":"2402.03300","n_code_links":5,"syntology":{"ran":14,"of":24,"n_ran_checked":10,"n_instrument":4,"unverified":10,"pointer_only":3,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 4 where Syntology's instrument failed) · 10 unverified","official":{"repos":["deepseek-ai/deepseek-math"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"brain-bayesian-reward-conditioned-amortized","title":"BRAIn: Bayesian Reward-conditioned Amortized Inference for natural language generation from feedback","date":"2024-02-04","arxiv_id":"2402.02479","n_code_links":0,"syntology":null},{"paper":null,"slug":"hybrid-prediction-integrated-planning-for","title":"Hybrid-Prediction Integrated Planning for Autonomous Driving","date":"2024-02-04","arxiv_id":"2402.02426","n_code_links":0,"syntology":null},{"paper":null,"slug":"parametric-task-map-elites","title":"Parametric-Task MAP-Elites","date":"2024-02-02","arxiv_id":"2402.01275","n_code_links":0,"syntology":null},{"paper":null,"slug":"carff-conditional-auto-encoded-radiance-field","title":"CARFF: Conditional Auto-encoded Radiance Field for 3D Scene Forecasting","date":"2024-01-31","arxiv_id":"2401.18075","n_code_links":0,"syntology":null},{"paper":"/paper/simple-policy-optimization","slug":"simple-policy-optimization","title":"Simple Policy Optimization","date":"2024-01-29","arxiv_id":"2401.16025","n_code_links":1,"syntology":null},{"paper":"/paper/true-knowledge-comes-from-practice-aligning","slug":"true-knowledge-comes-from-practice-aligning","title":"True Knowledge Comes from Practice: Aligning LLMs with Embodied Environments via Reinforcement Learning","date":"2024-01-25","arxiv_id":"2401.14151","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["weihaotan/twosome"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/linear-alignment-a-closed-form-solution-for","slug":"linear-alignment-a-closed-form-solution-for","title":"Linear Alignment: A Closed-form Solution for Aligning Human Preferences without Tuning and Feedback","date":"2024-01-21","arxiv_id":"2401.11458","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["wizardcoast/linear_alignment"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"360orb-slam-a-visual-slam-system-for","title":"360ORB-SLAM: A Visual SLAM System for Panoramic Images with Depth Completion Network","date":"2024-01-19","arxiv_id":"2401.10560","n_code_links":0,"syntology":null},{"paper":"/paper/langprop-a-code-optimization-framework-using","slug":"langprop-a-code-optimization-framework-using","title":"LangProp: A code optimization framework using Large Language Models applied to driving","date":"2024-01-18","arxiv_id":"2401.10314","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["shuishida/langprop"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/reft-reasoning-with-reinforced-fine-tuning","slug":"reft-reasoning-with-reinforced-fine-tuning","title":"ReFT: Reasoning with Reinforced Fine-Tuning","date":"2024-01-17","arxiv_id":"2401.08967","n_code_links":1,"syntology":{"ran":5,"of":14,"n_ran_checked":5,"n_instrument":0,"unverified":9,"pointer_only":13,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 9 unverified","official":{"repos":["lqtrung1998/mwp_reft"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":9,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"cppo-continual-learning-for-reinforcement","title":"CPPO: Continual Learning for Reinforcement Learning with Human Feedback","date":"2024-01-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"sum-throughput-maximization-in-multi-bd","title":"Sum Throughput Maximization in Multi-BD Symbiotic Radio NOMA Network Assisted by Active-STAR-RIS","date":"2024-01-16","arxiv_id":"2401.08301","n_code_links":0,"syntology":null},{"paper":null,"slug":"drlc-reinforcement-learning-with-dense","title":"Beyond Sparse Rewards: Enhancing Reinforcement Learning with Language Model Critique in Text Generation","date":"2024-01-14","arxiv_id":"2401.07382","n_code_links":0,"syntology":null},{"paper":"/paper/aquarium-a-comprehensive-framework-for","slug":"aquarium-a-comprehensive-framework-for","title":"Aquarium: A Comprehensive Framework for Exploring Predator-Prey Dynamics through Multi-Agent Reinforcement Learning Algorithms","date":"2024-01-13","arxiv_id":"2401.07056","n_code_links":1,"syntology":null},{"paper":"/paper/mapo-advancing-multilingual-reasoning-through","slug":"mapo-advancing-multilingual-reasoning-through","title":"MAPO: Advancing Multilingual Reasoning through Multilingual Alignment-as-Preference Optimization","date":"2024-01-12","arxiv_id":"2401.06838","n_code_links":1,"syntology":{"ran":6,"of":10,"n_ran_checked":5,"n_instrument":1,"unverified":4,"pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","official":{"repos":["njunlp/mapo"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"autonomous-navigation-of-tractor-trailer","title":"Autonomous Navigation of Tractor-Trailer Vehicles through Roundabout Intersections","date":"2024-01-10","arxiv_id":"2401.04980","n_code_links":0,"syntology":null},{"paper":null,"slug":"feedback-guided-autonomous-driving","title":"Feedback-Guided Autonomous Driving","date":"2024-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"beyond-pid-controllers-ppo-with-neuralized","title":"Beyond PID Controllers: PPO with Neuralized PID Policy for Proton Beam Intensity Control in Mu2e","date":"2023-12-28","arxiv_id":"2312.17372","n_code_links":0,"syntology":null},{"paper":null,"slug":"preference-as-reward-maximum-preference","title":"Preference as Reward, Maximum Preference Optimization with Importance Sampling","date":"2023-12-27","arxiv_id":"2312.16430","n_code_links":0,"syntology":null},{"paper":"/paper/some-things-are-more-cringe-than-others","slug":"some-things-are-more-cringe-than-others","title":"Some things are more CRINGE than others: Iterative Preference Optimization with the Pairwise Cringe Loss","date":"2023-12-27","arxiv_id":"2312.16682","n_code_links":1,"syntology":{"ran":5,"of":8,"n_ran_checked":5,"n_instrument":0,"unverified":3,"pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":null}},{"paper":null,"slug":"agent-based-modelling-for-continuously","title":"Agent based modelling for continuously varying supply chains","date":"2023-12-24","arxiv_id":"2312.15502","n_code_links":0,"syntology":null},{"paper":"/paper/drivelm-driving-with-graph-visual-question","slug":"drivelm-driving-with-graph-visual-question","title":"DriveLM: Driving with Graph Visual Question Answering","date":"2023-12-21","arxiv_id":"2312.14150","n_code_links":3,"syntology":null},{"paper":"/paper/realistic-rainy-weather-simulation-for-lidars","slug":"realistic-rainy-weather-simulation-for-lidars","title":"Realistic Rainy Weather Simulation for LiDARs in CARLA Simulator","date":"2023-12-20","arxiv_id":"2312.12772","n_code_links":1,"syntology":null},{"paper":"/paper/colored-noise-in-ppo-improved-exploration-and","slug":"colored-noise-in-ppo-improved-exploration-and","title":"Colored Noise in PPO: Improved Exploration and Performance through Correlated Action Sampling","date":"2023-12-18","arxiv_id":"2312.11091","n_code_links":1,"syntology":null},{"paper":"/paper/gibbs-sampling-from-human-feedback-a-provable","slug":"gibbs-sampling-from-human-feedback-a-provable","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","date":"2023-12-18","arxiv_id":"2312.11456","n_code_links":3,"syntology":null},{"paper":"/paper/drivemlm-aligning-multi-modal-large-language","slug":"drivemlm-aligning-multi-modal-large-language","title":"DriveMLM: Aligning Multi-Modal Large Language Models with Behavioral Planning States for Autonomous Driving","date":"2023-12-14","arxiv_id":"2312.09245","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["opengvlab/drivemlm"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/gradient-informed-proximal-policy-1","slug":"gradient-informed-proximal-policy-1","title":"Gradient Informed Proximal Policy Optimization","date":"2023-12-14","arxiv_id":"2312.08710","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["sonsang/gippo"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/math-shepherd-a-label-free-step-by-step","slug":"math-shepherd-a-label-free-step-by-step","title":"Math-Shepherd: Verify and Reinforce LLMs Step-by-step without Human Annotations","date":"2023-12-14","arxiv_id":"2312.08935","n_code_links":3,"syntology":null},{"paper":null,"slug":"challenges-of-yolo-series-for-object","title":"Challenges of YOLO Series for Object Detection in Extremely Heavy Rain: CALRA Simulator based Synthetic Evaluation Dataset","date":"2023-12-13","arxiv_id":"2312.07976","n_code_links":0,"syntology":null},{"paper":"/paper/the-effective-horizon-explains-deep-rl","slug":"the-effective-horizon-explains-deep-rl","title":"The Effective Horizon Explains Deep RL Performance in Stochastic Environments","date":"2023-12-13","arxiv_id":"2312.08369","n_code_links":1,"syntology":null},{"paper":null,"slug":"adaptive-proximal-policy-optimization-with","title":"A dynamical clipping approach with task feedback for Proximal Policy Optimization","date":"2023-12-12","arxiv_id":"2312.07624","n_code_links":0,"syntology":null},{"paper":null,"slug":"skyscenes-a-synthetic-dataset-for-aerial","title":"SkyScenes: A Synthetic Dataset for Aerial Scene Understanding","date":"2023-12-11","arxiv_id":"2312.06719","n_code_links":0,"syntology":null},{"paper":null,"slug":"graph-based-prediction-and-planning-policy","title":"Graph-based Prediction and Planning Policy Network (GP3Net) for scalable self-driving in dynamic environments using Deep Reinforcement Learning","date":"2023-12-10","arxiv_id":"2312.05784","n_code_links":0,"syntology":null},{"paper":"/paper/can-language-agents-be-alternatives-to-ppo-a","slug":"can-language-agents-be-alternatives-to-ppo-a","title":"Can language agents be alternatives to PPO? A Preliminary Empirical Study On OpenAI Gym","date":"2023-12-06","arxiv_id":"2312.03290","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-reliable-representation-with-bidirectional","title":"A Reliable Representation with Bidirectional Transition Model for Visual Reinforcement Learning Generalization","date":"2023-12-04","arxiv_id":"2312.01915","n_code_links":0,"syntology":null},{"paper":null,"slug":"data-efficient-deep-reinforcement-learning-2","title":"Data-efficient Deep Reinforcement Learning for Vehicle Trajectory Control","date":"2023-11-30","arxiv_id":"2311.18393","n_code_links":0,"syntology":null},{"paper":"/paper/epitester-testing-autonomous-vehicles-with","slug":"epitester-testing-autonomous-vehicles-with","title":"EpiTESTER: Testing Autonomous Vehicles with Epigenetic Algorithm and Attention Mechanism","date":"2023-11-30","arxiv_id":"2312.00207","n_code_links":1,"syntology":null},{"paper":null,"slug":"safe-reinforcement-learning-in-a-simulated","title":"Safe Reinforcement Learning in a Simulated Robotic Arm","date":"2023-11-28","arxiv_id":"2312.09468","n_code_links":0,"syntology":null},{"paper":"/paper/an-efficient-game-theoretic-planner-for","slug":"an-efficient-game-theoretic-planner-for","title":"Automated Lane Merging via Game Theory and Branch Model Predictive Control","date":"2023-11-25","arxiv_id":"2311.14916","n_code_links":1,"syntology":null},{"paper":null,"slug":"attacking-motion-planners-using-adversarial","title":"Attacking Motion Planners Using Adversarial Perception Errors","date":"2023-11-21","arxiv_id":"2311.12722","n_code_links":0,"syntology":null},{"paper":"/paper/nav-q-quantum-deep-reinforcement-learning-for","slug":"nav-q-quantum-deep-reinforcement-learning-for","title":"Nav-Q: Quantum Deep Reinforcement Learning for Collision-Free Navigation of Self-Driving Cars","date":"2023-11-20","arxiv_id":"2311.12875","n_code_links":1,"syntology":null},{"paper":null,"slug":"bridging-data-driven-and-knowledge-driven","title":"Bridging Data-Driven and Knowledge-Driven Approaches for Safety-Critical Scenario Generation in Automated Vehicle Validation","date":"2023-11-18","arxiv_id":"2311.10937","n_code_links":0,"syntology":null},{"paper":null,"slug":"automatic-generation-of-scenarios-for-system","title":"Automatic Generation of Scenarios for System-level Simulation-based Verification of Autonomous Driving Systems","date":"2023-11-16","arxiv_id":"2311.09784","n_code_links":0,"syntology":null}],"record_sha256":"f7ee73e2e5f50df0835ac8b2eeff8455bb6e2db8d2fe8072f7d39a7f38bdc947","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}