{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/entropy-regularization/papers/4","list_of":"/method/entropy-regularization","method":"Entropy Regularization","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":4,"pages_in_order":12,"rows_per_page":100,"rows":[301,400],"of":1128,"counts":{"archive_papers_tagged":1128,"with_a_code_link":451,"where_syntology_ran_a_sample":156,"not_listed_spam_title":0,"listed":1128,"listed_where_code_ran":156,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":129,"every_run_a_failure_of_syntologys_instrument":27,"listed_with_a_run_with_no_instrument_failure":129,"listed_every_run_a_failure_of_syntologys_instrument":27,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/entropy-regularization","prev":"/method/entropy-regularization/papers/3","next":"/method/entropy-regularization/papers/5","papers":[{"paper":"/paper/flow-of-reasoning-efficient-training-of-llm","slug":"flow-of-reasoning-efficient-training-of-llm","title":"Flow of Reasoning:Training LLMs for Divergent Problem Solving with Minimal Examples","date":"2024-06-09","arxiv_id":"2406.05673","n_code_links":1,"syntology":{"ran":11,"of":15,"n_ran_checked":11,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["yu-fangxu/for"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"online-policy-distillation-with-decision","title":"Online Policy Distillation with Decision-Attention","date":"2024-06-08","arxiv_id":"2406.05488","n_code_links":0,"syntology":null},{"paper":"/paper/bench2drive-towards-multi-ability","slug":"bench2drive-towards-multi-ability","title":"Bench2Drive: Towards Multi-Ability Benchmarking of Closed-Loop End-To-End Autonomous Driving","date":"2024-06-06","arxiv_id":"2406.03877","n_code_links":4,"syntology":{"ran":13,"of":17,"n_ran_checked":10,"n_instrument":3,"unverified":4,"pointer_only":17,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","official":{"repos":["Thinklab-SJTU/Bench2Drive","autonomousvision/carla_garage","thinklab-sjtu/bench2drivezoo"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"bisimulation-metrics-are-optimal-transport","title":"Bisimulation Metrics are Optimal Transport Distances, and Can be Computed Efficiently","date":"2024-06-06","arxiv_id":"2406.04056","n_code_links":0,"syntology":null},{"paper":null,"slug":"essentially-sharp-estimates-on-the-entropy","title":"Optimal Rates of Convergence for Entropy Regularization in Discounted Markov Decision Processes","date":"2024-06-06","arxiv_id":"2406.04163","n_code_links":0,"syntology":null},{"paper":"/paper/hackatari-atari-learning-environments-for","slug":"hackatari-atari-learning-environments-for","title":"HackAtari: Atari Learning Environments for Robust and Continual Reinforcement Learning","date":"2024-06-06","arxiv_id":"2406.03997","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["k4ntz/HackAtari"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"transductive-off-policy-proximal-policy","title":"Transductive Off-policy Proximal Policy Optimization","date":"2024-06-06","arxiv_id":"2406.03894","n_code_links":0,"syntology":null},{"paper":null,"slug":"prompt-based-visual-alignment-for-zero-shot","title":"Prompt-based Visual Alignment for Zero-shot Policy Transfer","date":"2024-06-05","arxiv_id":"2406.03250","n_code_links":0,"syntology":null},{"paper":null,"slug":"aligning-large-language-models-via-fine","title":"Aligning Large Language Models via Fine-grained Supervision","date":"2024-06-04","arxiv_id":"2406.02756","n_code_links":0,"syntology":null},{"paper":null,"slug":"lanevil-benchmarking-the-robustness-of-lane","title":"LanEvil: Benchmarking the Robustness of Lane Detection to Environmental Illusions","date":"2024-06-03","arxiv_id":"2406.00934","n_code_links":0,"syntology":null},{"paper":"/paper/learning-from-mistakes-a-weakly-supervised","slug":"learning-from-mistakes-a-weakly-supervised","title":"Validity Learning on Failures: Mitigating the Distribution Shift in Autonomous Vehicle Planning","date":"2024-06-03","arxiv_id":"2406.01544","n_code_links":0,"syntology":null},{"paper":null,"slug":"self-improving-robust-preference-optimization","title":"Self-Improving Robust Preference Optimization","date":"2024-06-03","arxiv_id":"2406.01660","n_code_links":0,"syntology":null},{"paper":null,"slug":"pedestrian-intention-prediction-in-adverse","title":"Pedestrian intention prediction in Adverse Weather Conditions with Spiking Neural Networks and Dynamic Vision Sensors","date":"2024-06-01","arxiv_id":"2406.00473","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-deep-reinforcement-learning-approach-for-20","title":"A Deep Reinforcement Learning Approach for Trading Optimization in the Forex Market with Multi-Agent Asynchronous Distribution","date":"2024-05-30","arxiv_id":"2405.19982","n_code_links":0,"syntology":null},{"paper":"/paper/aquatic-navigation-a-challenging-benchmark","slug":"aquatic-navigation-a-challenging-benchmark","title":"Aquatic Navigation: A Challenging Benchmark for Deep Reinforcement Learning","date":"2024-05-30","arxiv_id":"2405.20534","n_code_links":1,"syntology":null},{"paper":null,"slug":"entropy-annealing-for-policy-mirror-descent","title":"Entropy annealing for policy mirror descent in continuous time and space","date":"2024-05-30","arxiv_id":"2405.20250","n_code_links":0,"syntology":null},{"paper":"/paper/learning-task-relevant-sequence","slug":"learning-task-relevant-sequence","title":"Intrinsic Dynamics-Driven Generalizable Scene Representations for Vision-Oriented Decision-Making Applications","date":"2024-05-30","arxiv_id":"2405.19736","n_code_links":1,"syntology":null},{"paper":null,"slug":"linear-function-approximation-as-a","title":"Linear Function Approximation as a Computationally Efficient Method to Solve Classical Reinforcement Learning Challenges","date":"2024-05-27","arxiv_id":"2405.20350","n_code_links":0,"syntology":null},{"paper":null,"slug":"scarl-a-synthetic-multi-modal-dataset-for","title":"SCaRL- A Synthetic Multi-Modal Dataset for Autonomous Driving","date":"2024-05-27","arxiv_id":"2405.17030","n_code_links":0,"syntology":null},{"paper":"/paper/symmetric-reinforcement-learning-loss-for","slug":"symmetric-reinforcement-learning-loss-for","title":"Symmetric Reinforcement Learning Loss for Robust Learning on Diverse Tasks and Model Scales","date":"2024-05-27","arxiv_id":"2405.17618","n_code_links":1,"syntology":null},{"paper":"/paper/rewarded-region-replay-r3-for-policy-learning","slug":"rewarded-region-replay-r3-for-policy-learning","title":"Rewarded Region Replay (R3) for Policy Learning with Discrete Action Space","date":"2024-05-26","arxiv_id":"2405.16383","n_code_links":1,"syntology":null},{"paper":"/paper/diffusion-based-reinforcement-learning-via-q","slug":"diffusion-based-reinforcement-learning-via-q","title":"Diffusion-based Reinforcement Learning via Q-weighted Variational Policy Optimization","date":"2024-05-25","arxiv_id":"2405.16173","n_code_links":1,"syntology":null},{"paper":"/paper/continuously-learning-adapting-and-improving","slug":"continuously-learning-adapting-and-improving","title":"Continuously Learning, Adapting, and Improving: A Dual-Process Approach to Autonomous Driving","date":"2024-05-24","arxiv_id":"2405.15324","n_code_links":1,"syntology":null},{"paper":null,"slug":"efficient-reinforcement-learning-via-large","title":"Extracting Heuristics from Large Language Models for Reward Shaping in Reinforcement Learning","date":"2024-05-24","arxiv_id":"2405.15194","n_code_links":0,"syntology":null},{"paper":"/paper/agile-a-novel-framework-of-llm-agents","slug":"agile-a-novel-framework-of-llm-agents","title":"AGILE: A Novel Reinforcement Learning Framework of LLM Agents","date":"2024-05-23","arxiv_id":"2405.14751","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":3,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["bytarnish/agile"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/chatscene-knowledge-enabled-safety-critical","slug":"chatscene-knowledge-enabled-safety-critical","title":"ChatScene: Knowledge-Enabled Safety-Critical Scenario Generation for Autonomous Vehicles","date":"2024-05-22","arxiv_id":"2405.14062","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["javyduck/ChatScene"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"maskfuser-masked-fusion-of-joint-multi-modal","title":"MaskFuser: Masked Fusion of Joint Multi-Modal Tokenization for End-to-End Autonomous Driving","date":"2024-05-13","arxiv_id":"2405.07573","n_code_links":0,"syntology":null},{"paper":"/paper/value-augmented-sampling-for-language-model","slug":"value-augmented-sampling-for-language-model","title":"Value Augmented Sampling for Language Model Alignment and Personalization","date":"2024-05-10","arxiv_id":"2405.06639","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["idanshen/Value-Augmented-Sampling"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/proximal-policy-optimization-with-adaptive-1","slug":"proximal-policy-optimization-with-adaptive-1","title":"Proximal Policy Optimization with Adaptive Exploration","date":"2024-05-07","arxiv_id":"2405.04664","n_code_links":1,"syntology":null},{"paper":null,"slug":"guidance-design-for-escape-flight-vehicle","title":"Guidance Design for Escape Flight Vehicle Using Evolution Strategy Enhanced Deep Reinforcement Learning","date":"2024-05-04","arxiv_id":"2405.03711","n_code_links":0,"syntology":null},{"paper":"/paper/d2po-discriminator-guided-dpo-with-response","slug":"d2po-discriminator-guided-dpo-with-response","title":"D2PO: Discriminator-Guided DPO with Response Evaluation Models","date":"2024-05-02","arxiv_id":"2405.01511","n_code_links":1,"syntology":{"ran":11,"of":13,"n_ran_checked":11,"n_instrument":0,"unverified":2,"pointer_only":13,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["PrasannS/d2po"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/no-representation-no-trust-connecting","slug":"no-representation-no-trust-connecting","title":"No Representation, No Trust: Connecting Representation, Collapse, and Trust Issues in PPO","date":"2024-05-01","arxiv_id":"2405.00662","n_code_links":1,"syntology":{"ran":6,"of":8,"n_ran_checked":5,"n_instrument":1,"unverified":2,"pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["claire-labo/no-representation-no-trust"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/guiding-attention-in-end-to-end-driving","slug":"guiding-attention-in-end-to-end-driving","title":"Guiding Attention in End-to-End Driving Models","date":"2024-04-30","arxiv_id":"2405.00242","n_code_links":1,"syntology":null},{"paper":"/paper/dpo-meets-ppo-reinforced-token-optimization","slug":"dpo-meets-ppo-reinforced-token-optimization","title":"DPO Meets PPO: Reinforced Token Optimization for RLHF","date":"2024-04-29","arxiv_id":"2404.18922","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":3,"n_instrument":1,"unverified":2,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["zkshan2002/rto"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/3d-extended-object-tracking-by-fusing","slug":"3d-extended-object-tracking-by-fusing","title":"3D Extended Object Tracking by Fusing Roadside Sparse Radar Point Clouds and Pixel Keypoints","date":"2024-04-27","arxiv_id":"2404.17903","n_code_links":2,"syntology":null},{"paper":"/paper/rebel-reinforcement-learning-via-regressing","slug":"rebel-reinforcement-learning-via-regressing","title":"REBEL: Reinforcement Learning via Regressing Relative Rewards","date":"2024-04-25","arxiv_id":"2404.16767","n_code_links":3,"syntology":{"ran":16,"of":20,"n_ran_checked":12,"n_instrument":4,"unverified":4,"pointer_only":6,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 0 violated, 11 with no contract checked; 4 where Syntology's instrument failed) · 4 unverified","official":{"repos":["Owen-Oertell/rlcm","zhaolingao/rebel"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"contextualfusion-context-based-multi-sensor","title":"ContextualFusion: Context-Based Multi-Sensor Fusion for 3D Object Detection in Adverse Operating Conditions","date":"2024-04-23","arxiv_id":"2404.14780","n_code_links":0,"syntology":null},{"paper":"/paper/is-dpo-superior-to-ppo-for-llm-alignment-a","slug":"is-dpo-superior-to-ppo-for-llm-alignment-a","title":"Is DPO Superior to PPO for LLM Alignment? A Comprehensive Study","date":"2024-04-16","arxiv_id":"2404.10719","n_code_links":1,"syntology":null},{"paper":"/paper/joint-physical-digital-facial-attack","slug":"joint-physical-digital-facial-attack","title":"Joint Physical-Digital Facial Attack Detection Via Simulating Spoofing Clues","date":"2024-04-12","arxiv_id":"2404.08450","n_code_links":3,"syntology":null},{"paper":"/paper/sevd-synthetic-event-based-vision-dataset-for","slug":"sevd-synthetic-event-based-vision-dataset-for","title":"SEVD: Synthetic Event-based Vision Dataset for Ego and Fixed Traffic Perception","date":"2024-04-12","arxiv_id":"2404.10540","n_code_links":1,"syntology":null},{"paper":null,"slug":"enhanced-cooperative-perception-for","title":"Enhanced Cooperative Perception for Autonomous Vehicles Using Imperfect Communication","date":"2024-04-10","arxiv_id":"2404.08013","n_code_links":0,"syntology":null},{"paper":null,"slug":"synergy-of-large-language-model-and-model","title":"Synergy of Large Language Model and Model Driven Engineering for Automated Development of Centralized Vehicular Systems","date":"2024-04-08","arxiv_id":"2404.05508","n_code_links":0,"syntology":null},{"paper":null,"slug":"prompting-multi-modal-tokens-to-enhance-end","title":"Prompting Multi-Modal Tokens to Enhance End-to-End Autonomous Driving Imitation Learning with LLMs","date":"2024-04-07","arxiv_id":"2404.04869","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-proximal-policy-optimization-based","title":"A proximal policy optimization based intelligent home solar management","date":"2024-04-05","arxiv_id":"2404.03888","n_code_links":0,"syntology":null},{"paper":"/paper/agl-net-aerial-ground-cross-modal-global","slug":"agl-net-aerial-ground-cross-modal-global","title":"AGL-NET: Aerial-Ground Cross-Modal Global Localization with Varying Scales","date":"2024-04-04","arxiv_id":"2404.03187","n_code_links":1,"syntology":null},{"paper":"/paper/addressing-loss-of-plasticity-and","slug":"addressing-loss-of-plasticity-and","title":"Addressing Loss of Plasticity and Catastrophic Forgetting in Continual Learning","date":"2024-03-31","arxiv_id":"2404.00781","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["mohmdelsayed/upgd"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/human-compatible-driving-partners-through","slug":"human-compatible-driving-partners-through","title":"Human-compatible driving partners through data-regularized self-play reinforcement learning","date":"2024-03-28","arxiv_id":"2403.19648","n_code_links":1,"syntology":null},{"paper":"/paper/scenario-based-curriculum-generation-for","slug":"scenario-based-curriculum-generation-for","title":"Scenario-Based Curriculum Generation for Multi-Agent Autonomous Driving","date":"2024-03-26","arxiv_id":"2403.17805","n_code_links":1,"syntology":null},{"paper":null,"slug":"drivecot-integrating-chain-of-thought","title":"DriveCoT: Integrating Chain-of-Thought Reasoning with End-to-End Driving","date":"2024-03-25","arxiv_id":"2403.16996","n_code_links":0,"syntology":null},{"paper":null,"slug":"policy-optimization-finds-nash-equilibrium-in","title":"Policy Optimization finds Nash Equilibrium in Regularized General-Sum LQ Games","date":"2024-03-25","arxiv_id":"2404.00045","n_code_links":0,"syntology":null},{"paper":null,"slug":"predictable-interval-mdps-through-entropy","title":"Predictable Interval MDPs through Entropy Regularization","date":"2024-03-25","arxiv_id":"2403.16711","n_code_links":0,"syntology":null},{"paper":"/paper/policy-mirror-descent-with-lookahead","slug":"policy-mirror-descent-with-lookahead","title":"Policy Mirror Descent with Lookahead","date":"2024-03-21","arxiv_id":"2403.14156","n_code_links":1,"syntology":null},{"paper":"/paper/jaxued-a-simple-and-useable-ued-library-in","slug":"jaxued-a-simple-and-useable-ued-library-in","title":"JaxUED: A simple and useable UED library in Jax","date":"2024-03-19","arxiv_id":"2403.13091","n_code_links":1,"syntology":null},{"paper":null,"slug":"m2da-multi-modal-fusion-transformer","title":"M2DA: Multi-Modal Fusion Transformer Incorporating Driver Attention for Autonomous Driving","date":"2024-03-19","arxiv_id":"2403.12552","n_code_links":0,"syntology":null},{"paper":null,"slug":"adamer-ctc-connectionist-temporal","title":"AdaMER-CTC: Connectionist Temporal Classification with Adaptive Maximum Entropy Regularization for Automatic Speech Recognition","date":"2024-03-18","arxiv_id":"2403.11578","n_code_links":0,"syntology":null},{"paper":"/paper/driving-style-alignment-for-llm-powered","slug":"driving-style-alignment-for-llm-powered","title":"Driving Style Alignment for LLM-powered Driver Agent","date":"2024-03-17","arxiv_id":"2403.11368","n_code_links":2,"syntology":null},{"paper":null,"slug":"are-you-a-robot-detecting-autonomous-vehicles","title":"Are you a robot? Detecting Autonomous Vehicles from Behavior Analysis","date":"2024-03-14","arxiv_id":"2403.09571","n_code_links":0,"syntology":null},{"paper":null,"slug":"right-place-right-time-towards-objectnav-for","title":"Right Place, Right Time! Dynamizing Topological Graphs for Embodied Navigation","date":"2024-03-14","arxiv_id":"2403.09905","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-reinforcement-learning-from-human-1","title":"Improving Reinforcement Learning from Human Feedback Using Contrastive Rewards","date":"2024-03-12","arxiv_id":"2403.07708","n_code_links":0,"syntology":null},{"paper":null,"slug":"tractable-joint-prediction-and-planning-over","title":"Tractable Joint Prediction and Planning over Discrete Behavior Modes for Urban Driving","date":"2024-03-12","arxiv_id":"2403.07232","n_code_links":0,"syntology":null},{"paper":null,"slug":"risk-sensitive-rl-with-optimized-certainty","title":"Risk-Sensitive RL with Optimized Certainty Equivalents via Reduction to Standard RL","date":"2024-03-10","arxiv_id":"2403.06323","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-sinkhorn-type-algorithm-for-constrained","title":"A Sinkhorn-type Algorithm for Constrained Optimal Transport","date":"2024-03-08","arxiv_id":"2403.05054","n_code_links":0,"syntology":null},{"paper":null,"slug":"teaching-large-language-models-to-reason-with","title":"Teaching Large Language Models to Reason with Reinforcement Learning","date":"2024-03-07","arxiv_id":"2403.04642","n_code_links":0,"syntology":null},{"paper":null,"slug":"commit-certifying-robustness-of-multi-sensor","title":"COMMIT: Certifying Robustness of Multi-Sensor Fusion Systems against Semantic Attacks","date":"2024-03-04","arxiv_id":"2403.02329","n_code_links":0,"syntology":null},{"paper":null,"slug":"integrating-efficient-optimal-transport-and","title":"Integrating Efficient Optimal Transport and Functional Maps For Unsupervised Shape Correspondence Learning","date":"2024-03-04","arxiv_id":"2403.01781","n_code_links":0,"syntology":null},{"paper":null,"slug":"tsallis-entropy-regularization-for-linearly","title":"Tsallis Entropy Regularization for Linearly Solvable MDP and Linear Quadratic Regulator","date":"2024-03-04","arxiv_id":"2403.01805","n_code_links":0,"syntology":null},{"paper":"/paper/craftax-a-lightning-fast-benchmark-for-open","slug":"craftax-a-lightning-fast-benchmark-for-open","title":"Craftax: A Lightning-Fast Benchmark for Open-Ended Reinforcement Learning","date":"2024-02-26","arxiv_id":"2402.16801","n_code_links":1,"syntology":{"ran":2,"of":16,"n_ran_checked":2,"n_instrument":0,"unverified":14,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 14 unverified","official":{"repos":["michaeltmatthews/craftax"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":14,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"back-to-basics-revisiting-reinforce-style","title":"Back to Basics: Revisiting REINFORCE Style Optimization for Learning from Human Feedback in LLMs","date":"2024-02-22","arxiv_id":"2402.14740","n_code_links":0,"syntology":null},{"paper":null,"slug":"distributed-radiance-fields-for-edge-video","title":"Distributed Radiance Fields for Edge Video Compression and Metaverse Integration in Autonomous Driving","date":"2024-02-22","arxiv_id":"2402.14642","n_code_links":0,"syntology":null},{"paper":"/paper/hybrid-reasoning-based-on-large-language","slug":"hybrid-reasoning-based-on-large-language","title":"Hybrid Reasoning Based on Large Language Models for Autonomous Car Driving","date":"2024-02-21","arxiv_id":"2402.13602","n_code_links":1,"syntology":null},{"paper":"/paper/vadv2-end-to-end-vectorized-autonomous","slug":"vadv2-end-to-end-vectorized-autonomous","title":"VADv2: End-to-End Vectorized Autonomous Driving via Probabilistic Planning","date":"2024-02-20","arxiv_id":"2402.13243","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["hustvl/vad"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["community","official"]}}},{"paper":null,"slug":"surpassing-legacy-approaches-and-human","title":"Surpassing legacy approaches to PWR core reload optimization with single-objective Reinforcement learning","date":"2024-02-16","arxiv_id":"2402.11040","n_code_links":0,"syntology":null},{"paper":null,"slug":"rs-dpo-a-hybrid-rejection-sampling-and-direct","title":"RS-DPO: A Hybrid Rejection Sampling and Direct Preference Optimization Method for Alignment of Large Language Models","date":"2024-02-15","arxiv_id":"2402.10038","n_code_links":0,"syntology":null},{"paper":"/paper/entropy-regularized-point-based-value","slug":"entropy-regularized-point-based-value","title":"Entropy-regularized Point-based Value Iteration","date":"2024-02-14","arxiv_id":"2402.09388","n_code_links":1,"syntology":null},{"paper":"/paper/reducing-texture-bias-of-deep-neural-networks","slug":"reducing-texture-bias-of-deep-neural-networks","title":"Reducing Texture Bias of Deep Neural Networks via Edge Enhancing Diffusion","date":"2024-02-14","arxiv_id":"2402.09530","n_code_links":1,"syntology":null},{"paper":null,"slug":"semtra-a-semantic-skill-translator-for-cross","title":"SemTra: A Semantic Skill Translator for Cross-Domain Zero-Shot Policy Adaptation","date":"2024-02-12","arxiv_id":"2402.07418","n_code_links":0,"syntology":null},{"paper":"/paper/solving-deep-reinforcement-learning","slug":"solving-deep-reinforcement-learning","title":"Solving Deep Reinforcement Learning Tasks with Evolution Strategies and Linear Policy Networks","date":"2024-02-10","arxiv_id":"2402.06912","n_code_links":1,"syntology":null},{"paper":"/paper/entropy-regularized-token-level-policy","slug":"entropy-regularized-token-level-policy","title":"Entropy-Regularized Token-Level Policy Optimization for Language Agent Reinforcement","date":"2024-02-09","arxiv_id":"2402.06700","n_code_links":1,"syntology":{"ran":6,"of":9,"n_ran_checked":3,"n_instrument":3,"unverified":3,"pointer_only":9,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","official":{"repos":["morning9393/etpo"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"edge-caching-based-on-deep-reinforcement","title":"Attention-Enhanced Prioritized Proximal Policy Optimization for Adaptive Edge Caching","date":"2024-02-08","arxiv_id":"2402.14576","n_code_links":0,"syntology":null},{"paper":null,"slug":"convergence-for-natural-policy-gradient-on","title":"Convergence for Natural Policy Gradient on Infinite-State Queueing MDPs","date":"2024-02-07","arxiv_id":"2402.05274","n_code_links":0,"syntology":null},{"paper":null,"slug":"tuning-the-feedback-controller-gains-is-a","title":"Tuning the feedback controller gains is a simple way to improve autonomous driving performance","date":"2024-02-07","arxiv_id":"2402.05064","n_code_links":0,"syntology":null},{"paper":"/paper/compound-returns-reduce-variance-in","slug":"compound-returns-reduce-variance-in","title":"Averaging $n$-step Returns Reduces Variance in Reinforcement Learning","date":"2024-02-06","arxiv_id":"2402.03903","n_code_links":0,"syntology":{"ran":5,"of":8,"n_ran_checked":4,"n_instrument":1,"unverified":3,"pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":null}},{"paper":"/paper/learning-to-generate-explainable-stock","slug":"learning-to-generate-explainable-stock","title":"Learning to Generate Explainable Stock Predictions using Self-Reflective Large Language Models","date":"2024-02-06","arxiv_id":"2402.03659","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["koa-fin/sep"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/oasim-an-open-and-adaptive-simulator-based-on","slug":"oasim-an-open-and-adaptive-simulator-based-on","title":"OASim: an Open and Adaptive Simulator based on Neural Rendering for Autonomous Driving","date":"2024-02-06","arxiv_id":"2402.03830","n_code_links":1,"syntology":null},{"paper":"/paper/deepseekmath-pushing-the-limits-of","slug":"deepseekmath-pushing-the-limits-of","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","date":"2024-02-05","arxiv_id":"2402.03300","n_code_links":5,"syntology":{"ran":14,"of":24,"n_ran_checked":10,"n_instrument":4,"unverified":10,"pointer_only":3,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 4 where Syntology's instrument failed) · 10 unverified","official":{"repos":["deepseek-ai/deepseek-math"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"brain-bayesian-reward-conditioned-amortized","title":"BRAIn: Bayesian Reward-conditioned Amortized Inference for natural language generation from feedback","date":"2024-02-04","arxiv_id":"2402.02479","n_code_links":0,"syntology":null},{"paper":null,"slug":"hybrid-prediction-integrated-planning-for","title":"Hybrid-Prediction Integrated Planning for Autonomous Driving","date":"2024-02-04","arxiv_id":"2402.02426","n_code_links":0,"syntology":null},{"paper":null,"slug":"parametric-task-map-elites","title":"Parametric-Task MAP-Elites","date":"2024-02-02","arxiv_id":"2402.01275","n_code_links":0,"syntology":null},{"paper":null,"slug":"equivalence-of-the-empirical-risk","title":"Equivalence of the Empirical Risk Minimization to Regularization on the Family of f-Divergences","date":"2024-02-01","arxiv_id":"2402.00501","n_code_links":0,"syntology":null},{"paper":null,"slug":"carff-conditional-auto-encoded-radiance-field","title":"CARFF: Conditional Auto-encoded Radiance Field for 3D Scene Forecasting","date":"2024-01-31","arxiv_id":"2401.18075","n_code_links":0,"syntology":null},{"paper":"/paper/simple-policy-optimization","slug":"simple-policy-optimization","title":"Simple Policy Optimization","date":"2024-01-29","arxiv_id":"2401.16025","n_code_links":1,"syntology":null},{"paper":"/paper/true-knowledge-comes-from-practice-aligning","slug":"true-knowledge-comes-from-practice-aligning","title":"True Knowledge Comes from Practice: Aligning LLMs with Embodied Environments via Reinforcement Learning","date":"2024-01-25","arxiv_id":"2401.14151","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["weihaotan/twosome"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/linear-alignment-a-closed-form-solution-for","slug":"linear-alignment-a-closed-form-solution-for","title":"Linear Alignment: A Closed-form Solution for Aligning Human Preferences without Tuning and Feedback","date":"2024-01-21","arxiv_id":"2401.11458","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["wizardcoast/linear_alignment"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"360orb-slam-a-visual-slam-system-for","title":"360ORB-SLAM: A Visual SLAM System for Panoramic Images with Depth Completion Network","date":"2024-01-19","arxiv_id":"2401.10560","n_code_links":0,"syntology":null},{"paper":"/paper/langprop-a-code-optimization-framework-using","slug":"langprop-a-code-optimization-framework-using","title":"LangProp: A code optimization framework using Large Language Models applied to driving","date":"2024-01-18","arxiv_id":"2401.10314","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["shuishida/langprop"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/reft-reasoning-with-reinforced-fine-tuning","slug":"reft-reasoning-with-reinforced-fine-tuning","title":"ReFT: Reasoning with Reinforced Fine-Tuning","date":"2024-01-17","arxiv_id":"2401.08967","n_code_links":1,"syntology":{"ran":5,"of":14,"n_ran_checked":5,"n_instrument":0,"unverified":9,"pointer_only":13,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 9 unverified","official":{"repos":["lqtrung1998/mwp_reft"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":9,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"cppo-continual-learning-for-reinforcement","title":"CPPO: Continual Learning for Reinforcement Learning with Human Feedback","date":"2024-01-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"sum-throughput-maximization-in-multi-bd","title":"Sum Throughput Maximization in Multi-BD Symbiotic Radio NOMA Network Assisted by Active-STAR-RIS","date":"2024-01-16","arxiv_id":"2401.08301","n_code_links":0,"syntology":null},{"paper":null,"slug":"drlc-reinforcement-learning-with-dense","title":"Beyond Sparse Rewards: Enhancing Reinforcement Learning with Language Model Critique in Text Generation","date":"2024-01-14","arxiv_id":"2401.07382","n_code_links":0,"syntology":null},{"paper":"/paper/aquarium-a-comprehensive-framework-for","slug":"aquarium-a-comprehensive-framework-for","title":"Aquarium: A Comprehensive Framework for Exploring Predator-Prey Dynamics through Multi-Agent Reinforcement Learning Algorithms","date":"2024-01-13","arxiv_id":"2401.07056","n_code_links":1,"syntology":null}],"record_sha256":"5591039b6294d529d25e792efacacae9a82d77d4cb8a41118f60f83b9ba43278","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}