{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/entropy-regularization/papers/3","list_of":"/method/entropy-regularization","method":"Entropy Regularization","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":3,"pages_in_order":12,"rows_per_page":100,"rows":[201,300],"of":1128,"counts":{"archive_papers_tagged":1128,"with_a_code_link":451,"where_syntology_ran_a_sample":156,"not_listed_spam_title":0,"listed":1128,"listed_where_code_ran":156,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":129,"every_run_a_failure_of_syntologys_instrument":27,"listed_with_a_run_with_no_instrument_failure":129,"listed_every_run_a_failure_of_syntologys_instrument":27,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/entropy-regularization","prev":"/method/entropy-regularization/papers/2","next":"/method/entropy-regularization/papers/4","papers":[{"paper":"/paper/carla2real-a-tool-for-reducing-the-sim2real","slug":"carla2real-a-tool-for-reducing-the-sim2real","title":"CARLA2Real: a tool for reducing the sim2real gap in CARLA simulator","date":"2024-10-23","arxiv_id":"2410.18238","n_code_links":1,"syntology":null},{"paper":"/paper/exploring-rl-based-llm-training-for-formal","slug":"exploring-rl-based-llm-training-for-formal","title":"Exploring RL-based LLM Training for Formal Language Tasks with Programmed Rewards","date":"2024-10-22","arxiv_id":"2410.17126","n_code_links":1,"syntology":null},{"paper":null,"slug":"hierarchical-multi-agent-reinforcement-2","title":"Hierarchical Multi-agent Reinforcement Learning for Cyber Network Defense","date":"2024-10-22","arxiv_id":"2410.17351","n_code_links":0,"syntology":null},{"paper":null,"slug":"survival-of-the-fittest-evolutionary","title":"Survival of the Fittest: Evolutionary Adaptation of Policies for Environmental Shifts","date":"2024-10-22","arxiv_id":"2410.19852","n_code_links":0,"syntology":null},{"paper":null,"slug":"how-to-build-a-pre-trained-multimodal-model","title":"How to Build a Pre-trained Multimodal model for Simultaneously Chatting and Decision-making?","date":"2024-10-21","arxiv_id":"2410.15885","n_code_links":0,"syntology":null},{"paper":"/paper/benchmarking-deep-reinforcement-learning-for-1","slug":"benchmarking-deep-reinforcement-learning-for-1","title":"Benchmarking Deep Reinforcement Learning for Navigation in Denied Sensor Environments","date":"2024-10-18","arxiv_id":"2410.14616","n_code_links":1,"syntology":null},{"paper":null,"slug":"on-the-influence-of-shape-texture-and-color","title":"On the Influence of Shape, Texture and Color for Learning Semantic Segmentation","date":"2024-10-18","arxiv_id":"2410.14878","n_code_links":0,"syntology":null},{"paper":"/paper/unidrive-towards-universal-driving-perception","slug":"unidrive-towards-universal-driving-perception","title":"UniDrive: Towards Universal Driving Perception Across Camera Configurations","date":"2024-10-17","arxiv_id":"2410.13864","n_code_links":1,"syntology":{"ran":6,"of":8,"n_ran_checked":5,"n_instrument":1,"unverified":2,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["ywyeli/unidrive"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"aero-softmax-only-llms-for-efficient-private","title":"AERO: Softmax-Only LLMs for Efficient Private Inference","date":"2024-10-16","arxiv_id":"2410.13060","n_code_links":0,"syntology":null},{"paper":null,"slug":"physical-informed-inspired-deep-reinforcement","title":"Physical Informed-Inspired Deep Reinforcement Learning Based Bi-Level Programming for Microgrid Scheduling","date":"2024-10-15","arxiv_id":"2410.11932","n_code_links":0,"syntology":null},{"paper":null,"slug":"astm-autonomous-smart-traffic-management","title":"ASTM :Autonomous Smart Traffic Management System Using Artificial Intelligence CNN and LSTM","date":"2024-10-14","arxiv_id":"2410.10929","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-the-language-understanding","title":"Improving the Language Understanding Capabilities of Large Language Models Using Reinforcement Learning","date":"2024-10-14","arxiv_id":"2410.11020","n_code_links":0,"syntology":null},{"paper":"/paper/taming-overconfidence-in-llms-reward","slug":"taming-overconfidence-in-llms-reward","title":"Taming Overconfidence in LLMs: Reward Calibration in RLHF","date":"2024-10-13","arxiv_id":"2410.09724","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["SeanLeng1/Reward-Calibration"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"increasing-the-difficulty-of-automatically","title":"Increasing the Difficulty of Automatically Generated Questions via Reinforcement Learning with Synthetic Preference","date":"2024-10-10","arxiv_id":"2410.08289","n_code_links":0,"syntology":null},{"paper":"/paper/happy-a-debiased-learning-framework-for","slug":"happy-a-debiased-learning-framework-for","title":"Happy: A Debiased Learning Framework for Continual Generalized Category Discovery","date":"2024-10-09","arxiv_id":"2410.06535","n_code_links":1,"syntology":{"ran":9,"of":16,"n_ran_checked":3,"n_instrument":6,"unverified":7,"pointer_only":1,"phrase":"9 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 6 where Syntology's instrument failed) · 7 unverified","official":{"repos":["mashijie1028/happy-cgcd"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":7,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"adver-city-open-source-multi-modal-dataset","title":"Adver-City: Open-Source Multi-Modal Dataset for Collaborative Perception Under Adverse Weather Conditions","date":"2024-10-08","arxiv_id":"2410.06380","n_code_links":0,"syntology":null},{"paper":"/paper/coevolving-with-the-other-you-fine-tuning-llm","slug":"coevolving-with-the-other-you-fine-tuning-llm","title":"Coevolving with the Other You: Fine-Tuning LLM with Sequential Cooperative Multi-Agent Reinforcement Learning","date":"2024-10-08","arxiv_id":"2410.06101","n_code_links":1,"syntology":{"ran":8,"of":11,"n_ran_checked":6,"n_instrument":2,"unverified":3,"pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","official":{"repos":["Harry67Hu/CORY"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"human-in-the-loop-reasoning-for-traffic-sign","title":"Human-in-the-loop Reasoning For Traffic Sign Detection: Collaborative Approach Yolo With Video-llava","date":"2024-10-07","arxiv_id":"2410.05096","n_code_links":0,"syntology":null},{"paper":null,"slug":"mastering-chinese-chess-ai-xiangqi-without","title":"Mastering Chinese Chess AI (Xiangqi) Without Search","date":"2024-10-07","arxiv_id":"2410.04865","n_code_links":0,"syntology":null},{"paper":null,"slug":"implicit-to-explicit-entropy-regularization","title":"Implicit to Explicit Entropy Regularization: Benchmarking ViT Fine-tuning under Noisy Labels","date":"2024-10-05","arxiv_id":"2410.04256","n_code_links":0,"syntology":null},{"paper":null,"slug":"end-to-end-driving-in-high-interaction","title":"End-to-end Driving in High-Interaction Traffic Scenarios with Reinforcement Learning","date":"2024-10-03","arxiv_id":"2410.02253","n_code_links":0,"syntology":null},{"paper":null,"slug":"asymmetry-of-the-relative-entropy-in-the","title":"Asymmetry of the Relative Entropy in the Regularization of Empirical Risk Minimization","date":"2024-10-02","arxiv_id":"2410.02833","n_code_links":0,"syntology":null},{"paper":"/paper/vineppo-unlocking-rl-potential-for-llm","slug":"vineppo-unlocking-rl-potential-for-llm","title":"VinePPO: Unlocking RL Potential For LLM Reasoning Through Refined Credit Assignment","date":"2024-10-02","arxiv_id":"2410.01679","n_code_links":1,"syntology":{"ran":7,"of":9,"n_ran_checked":7,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["mcgill-nlp/vineppo"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"the-perfect-blend-redefining-rlhf-with","title":"The Perfect Blend: Redefining RLHF with Mixture of Judges","date":"2024-09-30","arxiv_id":"2409.20370","n_code_links":0,"syntology":null},{"paper":null,"slug":"generalizing-consistency-policy-to-visual-rl","title":"Generalizing Consistency Policy to Visual RL with Prioritized Proximal Experience Regularization","date":"2024-09-28","arxiv_id":"2410.00051","n_code_links":0,"syntology":null},{"paper":null,"slug":"spatial-reasoning-and-planning-for-deep","title":"Spatial Reasoning and Planning for Deep Embodied Agents","date":"2024-09-28","arxiv_id":"2409.19479","n_code_links":0,"syntology":null},{"paper":null,"slug":"criticality-and-safety-margins-for","title":"Criticality and Safety Margins for Reinforcement Learning","date":"2024-09-26","arxiv_id":"2409.18289","n_code_links":0,"syntology":null},{"paper":null,"slug":"good-data-is-all-imitation-learning-needs","title":"Good Data Is All Imitation Learning Needs","date":"2024-09-26","arxiv_id":"2409.17605","n_code_links":0,"syntology":null},{"paper":null,"slug":"navigation-in-a-simplified-urban-flow-through","title":"Navigation in a simplified Urban Flow through Deep Reinforcement Learning","date":"2024-09-26","arxiv_id":"2409.17922","n_code_links":0,"syntology":null},{"paper":"/paper/dashing-for-the-golden-snitch-multi-drone","slug":"dashing-for-the-golden-snitch-multi-drone","title":"Dashing for the Golden Snitch: Multi-Drone Time-Optimal Motion Planning with Multi-Agent Reinforcement Learning","date":"2024-09-25","arxiv_id":"2409.16720","n_code_links":1,"syntology":null},{"paper":null,"slug":"mitigating-covariate-shift-in-imitation-2","title":"Mitigating Covariate Shift in Imitation Learning for Autonomous Vehicles Using Latent Space Generative World Models","date":"2024-09-25","arxiv_id":"2409.16663","n_code_links":0,"syntology":null},{"paper":null,"slug":"safe-navigation-for-robotic-digestive","title":"Safe Navigation for Robotic Digestive Endoscopy via Human Intervention-based Reinforcement Learning","date":"2024-09-24","arxiv_id":"2409.15688","n_code_links":0,"syntology":null},{"paper":null,"slug":"quantifying-context-bias-in-domain-adaptation","title":"Quantifying Context Bias in Domain Adaptation for Object Detection","date":"2024-09-23","arxiv_id":"2409.14679","n_code_links":0,"syntology":null},{"paper":null,"slug":"lfp-efficient-and-accurate-end-to-end-lane","title":"LFP: Efficient and Accurate End-to-End Lane-Level Planning via Camera-LiDAR Fusion","date":"2024-09-21","arxiv_id":"2409.14170","n_code_links":0,"syntology":null},{"paper":null,"slug":"metdrive-multi-modal-end-to-end-autonomous","title":"METDrive: Multi-modal End-to-end Autonomous Driving with Temporal Guidance","date":"2024-09-19","arxiv_id":"2409.12667","n_code_links":0,"syntology":null},{"paper":null,"slug":"automating-proton-pbs-treatment-planning-for","title":"Automating proton PBS treatment planning for head and neck cancers using policy gradient-based deep reinforcement learning","date":"2024-09-17","arxiv_id":"2409.11576","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-policy-actor-critic-reinforcement-learning","title":"On-policy Actor-Critic Reinforcement Learning for Multi-UAV Exploration","date":"2024-09-17","arxiv_id":"2409.11058","n_code_links":0,"syntology":null},{"paper":null,"slug":"disentangling-uncertainty-for-safe-social","title":"Disentangling Uncertainty for Safe Social Navigation using Deep Reinforcement Learning","date":"2024-09-16","arxiv_id":"2409.10655","n_code_links":0,"syntology":null},{"paper":null,"slug":"risk-aware-autonomous-driving-for-linear","title":"Risk-Aware Autonomous Driving with Linear Temporal Logic Specifications","date":"2024-09-15","arxiv_id":"2409.09769","n_code_links":0,"syntology":null},{"paper":null,"slug":"applying-action-masking-and-curriculum","title":"Applying Action Masking and Curriculum Learning Techniques to Improve Data Efficiency and Overall Performance in Operational Technology Cyber Security using Reinforcement Learning","date":"2024-09-13","arxiv_id":"2409.10563","n_code_links":0,"syntology":null},{"paper":null,"slug":"module-wise-adaptive-adversarial-training-for","title":"Module-wise Adaptive Adversarial Training for End-to-end Autonomous Driving","date":"2024-09-11","arxiv_id":"2409.07321","n_code_links":0,"syntology":null},{"paper":"/paper/entaugment-entropy-driven-adaptive-data","slug":"entaugment-entropy-driven-adaptive-data","title":"EntAugment: Entropy-Driven Adaptive Data Augmentation Framework for Image Classification","date":"2024-09-10","arxiv_id":"2409.06290","n_code_links":1,"syntology":{"ran":7,"of":10,"n_ran_checked":4,"n_instrument":3,"unverified":3,"pointer_only":10,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","official":{"repos":["jackbrocp/entaugment"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/multi-v2x-a-large-scale-multi-modal-multi","slug":"multi-v2x-a-large-scale-multi-modal-multi","title":"Multi-V2X: A Large Scale Multi-modal Multi-penetration-rate Dataset for Cooperative Perception","date":"2024-09-08","arxiv_id":"2409.04980","n_code_links":1,"syntology":null},{"paper":null,"slug":"quantfactor-reinforce-mining-steady-formulaic","title":"QuantFactor REINFORCE: Mining Steady Formulaic Alpha Factors with Variance-bounded REINFORCE","date":"2024-09-08","arxiv_id":"2409.05144","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-enabled-satellite","title":"Reinforcement Learning-enabled Satellite Constellation Reconfiguration and Retasking for Mission-Critical Applications","date":"2024-09-03","arxiv_id":"2409.02270","n_code_links":0,"syntology":null},{"paper":null,"slug":"development-of-occupancy-prediction-algorithm","title":"Development of Occupancy Prediction Algorithm for Underground Parking Lots","date":"2024-09-02","arxiv_id":"2409.00923","n_code_links":0,"syntology":null},{"paper":"/paper/enhancing-sample-efficiency-and-exploration","slug":"enhancing-sample-efficiency-and-exploration","title":"Enhancing Sample Efficiency and Exploration in Reinforcement Learning through the Integration of Diffusion Models and Proximal Policy Optimization","date":"2024-09-02","arxiv_id":"2409.01427","n_code_links":1,"syntology":null},{"paper":"/paper/maferw-query-rewriting-with-multi-aspect","slug":"maferw-query-rewriting-with-multi-aspect","title":"MaFeRw: Query Rewriting with Multi-Aspect Feedbacks for Retrieval-Augmented Large Language Models","date":"2024-08-30","arxiv_id":"2408.17072","n_code_links":1,"syntology":null},{"paper":"/paper/towards-modality-agnostic-label-efficient","slug":"towards-modality-agnostic-label-efficient","title":"Towards Modality-agnostic Label-efficient Segmentation with Entropy-Regularized Distribution Alignment","date":"2024-08-29","arxiv_id":"2408.16520","n_code_links":1,"syntology":null},{"paper":null,"slug":"an-extremely-data-efficient-and-generative","title":"An Extremely Data-efficient and Generative LLM-based Reinforcement Learning Agent for Recommenders","date":"2024-08-28","arxiv_id":"2408.16032","n_code_links":0,"syntology":null},{"paper":null,"slug":"comparison-of-model-predictive-control-and","title":"Comparison of Model Predictive Control and Proximal Policy Optimization for a 1-DOF Helicopter System","date":"2024-08-28","arxiv_id":"2408.15633","n_code_links":0,"syntology":null},{"paper":"/paper/instruct-skillmix-a-powerful-pipeline-for-llm","slug":"instruct-skillmix-a-powerful-pipeline-for-llm","title":"Instruct-SkillMix: A Powerful Pipeline for LLM Instruction Tuning","date":"2024-08-27","arxiv_id":"2408.14774","n_code_links":1,"syntology":{"ran":7,"of":8,"n_ran_checked":6,"n_instrument":1,"unverified":1,"pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["princeton-pli/Instruct-SkillMix"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"inverse-q-token-level-reinforcement-learning","title":"Inverse-Q*: Token Level Reinforcement Learning for Aligning Large Language Models Without Preference Data","date":"2024-08-27","arxiv_id":"2408.14874","n_code_links":0,"syntology":null},{"paper":null,"slug":"pareto-inverse-reinforcement-learning-for","title":"Pareto Inverse Reinforcement Learning for Diverse Expert Policy Generation","date":"2024-08-22","arxiv_id":"2408.12110","n_code_links":0,"syntology":null},{"paper":null,"slug":"carla-drone-monocular-3d-object-detection","title":"CARLA Drone: Monocular 3D Object Detection from a Different Perspective","date":"2024-08-21","arxiv_id":"2408.11958","n_code_links":0,"syntology":null},{"paper":null,"slug":"minor-sft-loss-for-llm-fine-tune-to-increase","title":"Minor SFT loss for LLM fine-tune to increase performance and reduce model deviation","date":"2024-08-20","arxiv_id":"2408.10642","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploratory-optimal-stopping-a-singular","title":"Exploratory Optimal Stopping: A Singular Control Formulation","date":"2024-08-18","arxiv_id":"2408.09335","n_code_links":0,"syntology":null},{"paper":"/paper/syntrac-a-synthetic-dataset-for-traffic","slug":"syntrac-a-synthetic-dataset-for-traffic","title":"SynTraC: A Synthetic Dataset for Traffic Signal Control from Traffic Monitoring Cameras","date":"2024-08-18","arxiv_id":"2408.09588","n_code_links":1,"syntology":null},{"paper":"/paper/s-raf-a-simulation-based-robustness","slug":"s-raf-a-simulation-based-robustness","title":"S-RAF: A Simulation-Based Robustness Assessment Framework for Responsible Autonomous Driving","date":"2024-08-16","arxiv_id":"2408.08584","n_code_links":1,"syntology":null},{"paper":"/paper/enhancing-autonomous-vehicle-perception-in","slug":"enhancing-autonomous-vehicle-perception-in","title":"Enhancing Autonomous Vehicle Perception in Adverse Weather through Image Augmentation during Semantic Segmentation Training","date":"2024-08-14","arxiv_id":"2408.07239","n_code_links":1,"syntology":null},{"paper":"/paper/kolmogorov-arnold-network-for-online","slug":"kolmogorov-arnold-network-for-online","title":"Kolmogorov-Arnold Network for Online Reinforcement Learning","date":"2024-08-09","arxiv_id":"2408.04841","n_code_links":1,"syntology":null},{"paper":null,"slug":"2408-03003","title":"Cross-cultural analysis of pedestrian group behaviour influence on crossing decisions in interactions with autonomous vehicles","date":"2024-08-06","arxiv_id":"2408.03003","n_code_links":0,"syntology":null},{"paper":null,"slug":"2408-03084","title":"Research on Autonomous Driving Decision-making Strategies based Deep Reinforcement Learning","date":"2024-08-06","arxiv_id":"2408.03084","n_code_links":0,"syntology":null},{"paper":"/paper/2407-21310","slug":"2407-21310","title":"MSMA: Multi-agent Trajectory Prediction in Connected and Autonomous Vehicle Environment with Multi-source Data Integration","date":"2024-07-31","arxiv_id":"2407.21310","n_code_links":1,"syntology":null},{"paper":null,"slug":"appraisal-guided-proximal-policy-optimization","title":"Appraisal-Guided Proximal Policy Optimization: Modeling Psychological Disorders in Dynamic Grid World","date":"2024-07-29","arxiv_id":"2407.20383","n_code_links":0,"syntology":null},{"paper":null,"slug":"sapg-split-and-aggregate-policy-gradients","title":"SAPG: Split and Aggregate Policy Gradients","date":"2024-07-29","arxiv_id":"2407.20230","n_code_links":0,"syntology":null},{"paper":"/paper/towards-aligning-language-models-with-textual","slug":"towards-aligning-language-models-with-textual","title":"Towards Aligning Language Models with Textual Feedback","date":"2024-07-24","arxiv_id":"2407.16970","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["sauc-abadal/alt"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"in-search-for-architectures-and-loss","title":"In Search for Architectures and Loss Functions in Multi-Objective Reinforcement Learning","date":"2024-07-23","arxiv_id":"2407.16807","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluation-of-reinforcement-learning-for","title":"Evaluation of Reinforcement Learning for Autonomous Penetration Testing using A3C, Q-learning and DQN","date":"2024-07-22","arxiv_id":"2407.15656","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-hardware-fault-tolerance-in","title":"Enhancing Hardware Fault Tolerance in Machines with Reinforcement Learning Policy Gradient Algorithms","date":"2024-07-21","arxiv_id":"2407.15283","n_code_links":0,"syntology":null},{"paper":"/paper/understanding-reinforcement-learning-based","slug":"understanding-reinforcement-learning-based","title":"Understanding Reinforcement Learning-Based Fine-Tuning of Diffusion Models: A Tutorial and Review","date":"2024-07-18","arxiv_id":"2407.13734","n_code_links":1,"syntology":null},{"paper":null,"slug":"perception-helps-planning-facilitating-multi","title":"Perception Helps Planning: Facilitating Multi-Stage Lane-Level Integration via Double-Edge Structures","date":"2024-07-16","arxiv_id":"2407.11644","n_code_links":0,"syntology":null},{"paper":null,"slug":"dino-pre-training-for-vision-based-end-to-end","title":"DINO Pre-training for Vision-based End-to-end Autonomous Driving","date":"2024-07-15","arxiv_id":"2407.10803","n_code_links":0,"syntology":null},{"paper":null,"slug":"digital-twins-to-alleviate-the-need-for-real","title":"Digital twins to alleviate the need for real field data in vision-based vehicle speed detection systems","date":"2024-07-11","arxiv_id":"2407.08380","n_code_links":0,"syntology":null},{"paper":null,"slug":"real-time-system-optimal-traffic-routing","title":"Real-time system optimal traffic routing under uncertainties -- Can physics models boost reinforcement learning?","date":"2024-07-10","arxiv_id":"2407.07364","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-human-like-driving-active-inference","title":"Towards Human-Like Driving: Active Inference in Autonomous Vehicle Control","date":"2024-07-10","arxiv_id":"2407.07684","n_code_links":0,"syntology":null},{"paper":"/paper/exploring-the-causality-of-end-to-end","slug":"exploring-the-causality-of-end-to-end","title":"Exploring the Causality of End-to-End Autonomous Driving","date":"2024-07-09","arxiv_id":"2407.06546","n_code_links":1,"syntology":null},{"paper":null,"slug":"exposing-privacy-gaps-membership-inference","title":"Exposing Privacy Gaps: Membership Inference Attack on Preference Data for LLM Alignment","date":"2024-07-08","arxiv_id":"2407.06443","n_code_links":0,"syntology":null},{"paper":"/paper/simplifying-deep-temporal-difference-learning","slug":"simplifying-deep-temporal-difference-learning","title":"Simplifying Deep Temporal Difference Learning","date":"2024-07-05","arxiv_id":"2407.04811","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["mttga/purejaxql"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"continuous-time-q-learning-for-jump-diffusion","title":"Continuous-time q-Learning for Jump-Diffusion Models under Tsallis Entropy","date":"2024-07-04","arxiv_id":"2407.03888","n_code_links":0,"syntology":null},{"paper":null,"slug":"efficient-fusion-and-task-guided-embedding","title":"Efficient Fusion and Task Guided Embedding for End-to-end Autonomous Driving","date":"2024-07-03","arxiv_id":"2407.02878","n_code_links":0,"syntology":null},{"paper":null,"slug":"ppo-based-dynamic-control-of-uncertain","title":"PPO-based Dynamic Control of Uncertain Floating Platforms in the Zero-G Environment","date":"2024-07-03","arxiv_id":"2407.03224","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-deep-reinforcement-learning-approach-to-8","title":"A Deep Reinforcement Learning Approach to Battery Management in Dairy Farming via Proximal Policy Optimization","date":"2024-07-01","arxiv_id":"2407.01653","n_code_links":0,"syntology":null},{"paper":null,"slug":"acceleration-method-for-generating-perception","title":"Acceleration method for generating perception failure scenarios based on editing Markov process","date":"2024-07-01","arxiv_id":"2407.00980","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-for-adverse","title":"Deep Reinforcement Learning for Adverse Garage Scenario Generation","date":"2024-07-01","arxiv_id":"2407.01333","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-strategies-in","title":"Deep Reinforcement Learning Strategies in Finance: Insights into Asset Holding, Trading Behavior, and Purchase Diversity","date":"2024-06-29","arxiv_id":"2407.09557","n_code_links":0,"syntology":null},{"paper":"/paper/decoding-time-language-model-alignment-with","slug":"decoding-time-language-model-alignment-with","title":"Decoding-Time Language Model Alignment with Multiple Objectives","date":"2024-06-27","arxiv_id":"2406.18853","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["srzer/mod"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"end-to-end-autonomous-driving-without-costly","title":"End-to-End Autonomous Driving without Costly Modularization and 3D Manual Annotation","date":"2024-06-25","arxiv_id":"2406.17680","n_code_links":0,"syntology":null},{"paper":null,"slug":"performance-comparison-of-deep-rl-algorithms-1","title":"Performance Comparison of Deep RL Algorithms for Mixed Traffic Cooperative Lane-Changing","date":"2024-06-25","arxiv_id":"2407.02521","n_code_links":0,"syntology":null},{"paper":null,"slug":"racil-ray-tracing-based-multi-uav-obstacle","title":"RaCIL: Ray Tracing based Multi-UAV Obstacle Avoidance through Composite Imitation Learning","date":"2024-06-24","arxiv_id":"2407.02520","n_code_links":0,"syntology":null},{"paper":null,"slug":"multistep-criticality-search-and-power","title":"Multistep Criticality Search and Power Shaping in Microreactors with Reinforcement Learning","date":"2024-06-22","arxiv_id":"2406.15931","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-biomedical-knowledge-retrieval","title":"SeRTS: Self-Rewarding Tree Search for Biomedical Retrieval-Augmented Generation","date":"2024-06-17","arxiv_id":"2406.11258","n_code_links":0,"syntology":null},{"paper":null,"slug":"p-ta-using-proximal-policy-optimization-to","title":"P-TA: Using Proximal Policy Optimization to Enhance Tabular Data Augmentation via Large Language Models","date":"2024-06-17","arxiv_id":"2406.11391","n_code_links":0,"syntology":null},{"paper":null,"slug":"planning-with-adaptive-world-models-for","title":"Planning with Adaptive World Models for Autonomous Driving","date":"2024-06-15","arxiv_id":"2406.10714","n_code_links":0,"syntology":null},{"paper":"/paper/carllava-vision-language-models-for-camera","slug":"carllava-vision-language-models-for-camera","title":"CarLLaVA: Vision language models for camera-only closed-loop driving","date":"2024-06-14","arxiv_id":"2406.10165","n_code_links":1,"syntology":null},{"paper":null,"slug":"latent-assistance-networks-rediscovering","title":"Hadamard Representations: Augmenting Hyperbolic Tangents in RL","date":"2024-06-13","arxiv_id":"2406.09079","n_code_links":0,"syntology":null},{"paper":"/paper/unpacking-dpo-and-ppo-disentangling-best","slug":"unpacking-dpo-and-ppo-disentangling-best","title":"Unpacking DPO and PPO: Disentangling Best Practices for Learning from Preference Feedback","date":"2024-06-13","arxiv_id":"2406.09279","n_code_links":2,"syntology":{"ran":15,"of":19,"n_ran_checked":8,"n_instrument":7,"unverified":4,"pointer_only":0,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 7 where Syntology's instrument failed) · 4 unverified","official":{"repos":["allenai/open-instruct","hamishivi/easylm"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"optimizing-deep-reinforcement-learning-for-1","title":"Optimizing Deep Reinforcement Learning for Adaptive Robotic Arm Control","date":"2024-06-12","arxiv_id":"2407.02503","n_code_links":0,"syntology":null},{"paper":"/paper/priboot-a-new-data-driven-expert-for-improved","slug":"priboot-a-new-data-driven-expert-for-improved","title":"PRIBOOT: A New Data-Driven Expert for Improved Driving Simulations","date":"2024-06-12","arxiv_id":"2406.08421","n_code_links":1,"syntology":null},{"paper":null,"slug":"adaptive-opponent-policy-detection-in-multi","title":"Adaptive Opponent Policy Detection in Multi-Agent MDPs: Real-Time Strategy Switch Identification Using Running Error Estimation","date":"2024-06-10","arxiv_id":"2406.06500","n_code_links":0,"syntology":null}],"record_sha256":"d096c0b4ee7a730fc769631b96d1fcb4a279e52c9a32728d1e892936b789afeb","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}