{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/entropy-regularization/papers/6","list_of":"/method/entropy-regularization","method":"Entropy Regularization","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":6,"pages_in_order":12,"rows_per_page":100,"rows":[501,600],"of":1128,"counts":{"archive_papers_tagged":1128,"with_a_code_link":451,"where_syntology_ran_a_sample":156,"not_listed_spam_title":0,"listed":1128,"listed_where_code_ran":156,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":129,"every_run_a_failure_of_syntologys_instrument":27,"listed_with_a_run_with_no_instrument_failure":129,"listed_every_run_a_failure_of_syntologys_instrument":27,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/entropy-regularization","prev":"/method/entropy-regularization/papers/5","next":"/method/entropy-regularization/papers/7","papers":[{"paper":"/paper/learning-to-generate-better-than-your-llm","slug":"learning-to-generate-better-than-your-llm","title":"Learning to Generate Better Than Your LLM","date":"2023-06-20","arxiv_id":"2306.11816","n_code_links":1,"syntology":null},{"paper":null,"slug":"safe-efficient-comfort-and-energy-saving","title":"Safe, Efficient, Comfort, and Energy-saving Automated Driving through Roundabout Based on Deep Reinforcement Learning","date":"2023-06-20","arxiv_id":"2306.11465","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-study-on-quantifying-sim2real-image-gap-in","title":"A Study on Quantifying Sim2Real Image Gap in Autonomous Driving Simulations Using Lane Segmentation Attention Map Similarity","date":"2023-06-18","arxiv_id":"2306.10491","n_code_links":0,"syntology":null},{"paper":"/paper/coaching-a-teachable-student-1","slug":"coaching-a-teachable-student-1","title":"Coaching a Teachable Student","date":"2023-06-16","arxiv_id":"2306.10014","n_code_links":1,"syntology":{"ran":0,"of":4,"n_ran_checked":0,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"0 ran · 4 unverified","official":{"repos":["h2xlab/CaT"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":[]}}},{"paper":"/paper/hidden-biases-of-end-to-end-driving-models","slug":"hidden-biases-of-end-to-end-driving-models","title":"Hidden Biases of End-to-End Driving Models","date":"2023-06-13","arxiv_id":"2306.07957","n_code_links":1,"syntology":{"ran":13,"of":17,"n_ran_checked":10,"n_instrument":3,"unverified":4,"pointer_only":0,"phrase":"13 ran (of which 10 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","official":{"repos":["autonomousvision/carla_garage"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":10,"n_ran_no_instrument_failure":10,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"analysis-of-the-relative-entropy-asymmetry-in","title":"Analysis of the Relative Entropy Asymmetry in the Regularization of Empirical Risk Minimization","date":"2023-06-12","arxiv_id":"2306.07123","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-topic-extraction-in-recommender","title":"Enhancing Topic Extraction in Recommender Systems with Entropy Regularization","date":"2023-06-12","arxiv_id":"2306.07403","n_code_links":0,"syntology":null},{"paper":"/paper/backproptools-a-fast-portable-deep","slug":"backproptools-a-fast-portable-deep","title":"RLtools: A Fast, Portable Deep Reinforcement Learning Library for Continuous Control","date":"2023-06-06","arxiv_id":"2306.03530","n_code_links":1,"syntology":null},{"paper":"/paper/fine-tuning-language-models-with-advantage","slug":"fine-tuning-language-models-with-advantage","title":"Fine-Tuning Language Models with Advantage-Induced Policy Alignment","date":"2023-06-04","arxiv_id":"2306.02231","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":5,"n_instrument":0,"unverified":2,"pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["microsoft/rlhf-apa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"deep-q-learning-versus-proximal-policy","title":"Deep Q-Learning versus Proximal Policy Optimization: Performance Comparison in a Material Sorting Task","date":"2023-06-02","arxiv_id":"2306.01451","n_code_links":0,"syntology":null},{"paper":"/paper/relu-to-the-rescue-improve-your-on-policy","slug":"relu-to-the-rescue-improve-your-on-policy","title":"ReLU to the Rescue: Improve Your On-Policy Actor-Critic with Positive Advantages","date":"2023-06-02","arxiv_id":"2306.01460","n_code_links":1,"syntology":null},{"paper":"/paper/identifiability-and-generalizability-in","slug":"identifiability-and-generalizability-in","title":"Identifiability and Generalizability in Constrained Inverse Reinforcement Learning","date":"2023-06-01","arxiv_id":"2306.00629","n_code_links":1,"syntology":{"ran":11,"of":13,"n_ran_checked":11,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["andrschl/cirl"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/normalization-enhances-generalization-in","slug":"normalization-enhances-generalization-in","title":"Normalization Enhances Generalization in Visual Reinforcement Learning","date":"2023-06-01","arxiv_id":"2306.00656","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":0,"n_instrument":2,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["lilucse/Normalization-Enhances-Generalization-in-Visual-Reinforcement-Learning"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/latent-exploration-for-reinforcement-learning","slug":"latent-exploration-for-reinforcement-learning","title":"Latent Exploration for Reinforcement Learning","date":"2023-05-31","arxiv_id":"2305.20065","n_code_links":1,"syntology":null},{"paper":"/paper/exploring-the-promise-and-limits-of-real-time","slug":"exploring-the-promise-and-limits-of-real-time","title":"Exploring the Promise and Limits of Real-Time Recurrent Learning","date":"2023-05-30","arxiv_id":"2305.19044","n_code_links":1,"syntology":null},{"paper":null,"slug":"domo-ac-doubly-multi-step-off-policy-actor","title":"DoMo-AC: Doubly Multi-step Off-policy Actor-Critic Algorithm","date":"2023-05-29","arxiv_id":"2305.18501","n_code_links":0,"syntology":null},{"paper":null,"slug":"resilience-in-platoons-of-cooperative","title":"Resilience in Platoons of Cooperative Heterogeneous Vehicles: Self-organization Strategies and Provably-correct Design","date":"2023-05-27","arxiv_id":"2305.17443","n_code_links":0,"syntology":null},{"paper":"/paper/all-points-matter-entropy-regularized-1","slug":"all-points-matter-entropy-regularized-1","title":"All Points Matter: Entropy-Regularized Distribution Alignment for Weakly-supervised 3D Segmentation","date":"2023-05-25","arxiv_id":"2305.15832","n_code_links":1,"syntology":{"ran":16,"of":33,"n_ran_checked":3,"n_instrument":13,"unverified":17,"pointer_only":0,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 1 violated, 0 with no contract checked; 13 where Syntology's instrument failed) · 17 unverified","official":{"repos":["LiyaoTang/ERDA"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":17,"ran_from_kinds":["official"]}}},{"paper":"/paper/generating-synergistic-formulaic-alpha","slug":"generating-synergistic-formulaic-alpha","title":"Generating Synergistic Formulaic Alpha Collections via Reinforcement Learning","date":"2023-05-25","arxiv_id":"2306.12964","n_code_links":1,"syntology":null},{"paper":"/paper/realistically-distributing-object-placements","slug":"realistically-distributing-object-placements","title":"Realistically distributing object placements in synthetic training data improves the performance of vision-based object detection models","date":"2023-05-24","arxiv_id":"2305.14621","n_code_links":1,"syntology":null},{"paper":"/paper/alpacafarm-a-simulation-framework-for-methods-1","slug":"alpacafarm-a-simulation-framework-for-methods-1","title":"AlpacaFarm: A Simulation Framework for Methods that Learn from Human Feedback","date":"2023-05-22","arxiv_id":"2305.14387","n_code_links":2,"syntology":null},{"paper":null,"slug":"learning-pedestrian-actions-to-ensure-safe","title":"Learning Pedestrian Actions to Ensure Safe Autonomous Driving","date":"2023-05-22","arxiv_id":"2305.13051","n_code_links":0,"syntology":null},{"paper":"/paper/actor-critic-methods-using-physics-informed","slug":"actor-critic-methods-using-physics-informed","title":"Actor-Critic Methods using Physics-Informed Neural Networks: Control of a 1D PDE Model for Fluid-Cooled Battery Packs","date":"2023-05-18","arxiv_id":"2305.10952","n_code_links":1,"syntology":null},{"paper":"/paper/sharing-lifelong-reinforcement-learning","slug":"sharing-lifelong-reinforcement-learning","title":"Sharing Lifelong Reinforcement Learning Knowledge via Modulating Masks","date":"2023-05-18","arxiv_id":"2305.10997","n_code_links":2,"syntology":null},{"paper":"/paper/reasonnet-end-to-end-driving-with-temporal-1","slug":"reasonnet-end-to-end-driving-with-temporal-1","title":"ReasonNet: End-to-End Driving with Temporal and Global Reasoning","date":"2023-05-17","arxiv_id":"2305.10507","n_code_links":0,"syntology":null},{"paper":null,"slug":"slic-hf-sequence-likelihood-calibration-with","title":"SLiC-HF: Sequence Likelihood Calibration with Human Feedback","date":"2023-05-17","arxiv_id":"2305.10425","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-theoretical-analysis-of-optimistic-proximal","title":"A Theoretical Analysis of Optimistic Proximal Policy Optimization in Linear Markov Decision Processes","date":"2023-05-15","arxiv_id":"2305.08841","n_code_links":0,"syntology":null},{"paper":null,"slug":"dynamically-conservative-self-driving-planner","title":"Dynamically Conservative Self-Driving Planner for Long-Tail Cases","date":"2023-05-12","arxiv_id":"2305.07497","n_code_links":0,"syntology":null},{"paper":null,"slug":"policy-gradient-algorithms-implicitly","title":"Policy Gradient Algorithms Implicitly Optimize by Continuation","date":"2023-05-11","arxiv_id":"2305.06851","n_code_links":0,"syntology":null},{"paper":"/paper/think-twice-before-driving-towards-scalable","slug":"think-twice-before-driving-towards-scalable","title":"Think Twice before Driving: Towards Scalable Decoders for End-to-End Autonomous Driving","date":"2023-05-10","arxiv_id":"2305.06242","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["opendrivelab/thinktwice"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/reducing-the-cost-of-cycle-time-tuning-for","slug":"reducing-the-cost-of-cycle-time-tuning-for","title":"Reducing the Cost of Cycle-Time Tuning for Real-World Policy Optimization","date":"2023-05-09","arxiv_id":"2305.05760","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["homayoonfarrahi/cycle-time-study"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/local-optimization-achieves-global-optimality","slug":"local-optimization-achieves-global-optimality","title":"Local Optimization Achieves Global Optimality in Multi-Agent Reinforcement Learning","date":"2023-05-08","arxiv_id":"2305.04819","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["zhaoyl18/ratio_game"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/carla-bsp-a-simulated-dataset-with","slug":"carla-bsp-a-simulated-dataset-with","title":"CARLA-BSP: a simulated dataset with pedestrians","date":"2023-04-29","arxiv_id":"2305.00204","n_code_links":2,"syntology":null},{"paper":null,"slug":"adversarial-policy-optimization-in-deep","title":"Adversarial Policy Optimization in Deep Reinforcement Learning","date":"2023-04-27","arxiv_id":"2304.14533","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-agents-run-relay-race-with-strangers","title":"Can Agents Run Relay Race with Strangers? Generalization of RL to Out-of-Distribution Trajectories","date":"2023-04-26","arxiv_id":"2304.13424","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-end-to-end-vehicle-trajcetory-prediction","title":"An End-to-End Vehicle Trajcetory Prediction Framework","date":"2023-04-19","arxiv_id":"2304.09764","n_code_links":0,"syntology":null},{"paper":"/paper/bridging-rl-theory-and-practice-with-the-1","slug":"bridging-rl-theory-and-practice-with-the-1","title":"Bridging RL Theory and Practice with the Effective Horizon","date":"2023-04-19","arxiv_id":"2304.09853","n_code_links":1,"syntology":null},{"paper":null,"slug":"benchmarking-the-physical-world-adversarial","title":"Benchmarking the Physical-world Adversarial Robustness of Vehicle Detection","date":"2023-04-11","arxiv_id":"2304.05098","n_code_links":0,"syntology":null},{"paper":"/paper/rrhf-rank-responses-to-align-language-models","slug":"rrhf-rank-responses-to-align-language-models","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","date":"2023-04-11","arxiv_id":"2304.05302","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ganjinzero/rrhf"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"lane-lighting-aware-neural-fields-for","title":"LANe: Lighting-Aware Neural Fields for Compositional Scene Synthesis","date":"2023-04-06","arxiv_id":"2304.03280","n_code_links":0,"syntology":null},{"paper":"/paper/autorl-hyperparameter-landscapes","slug":"autorl-hyperparameter-landscapes","title":"AutoRL Hyperparameter Landscapes","date":"2023-04-05","arxiv_id":"2304.02396","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["automl/autorl-landscape"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"pac-based-formal-verification-for-out-of","title":"PAC-Based Formal Verification for Out-of-Distribution Data Detection","date":"2023-04-04","arxiv_id":"2304.01592","n_code_links":0,"syntology":null},{"paper":null,"slug":"understanding-reinforcement-learning","title":"Understanding Reinforcement Learning Algorithms: The Progress from Basic Q-learning to Proximal Policy Optimization","date":"2023-03-31","arxiv_id":"2304.00026","n_code_links":0,"syntology":null},{"paper":null,"slug":"specification-guided-data-aggregation-for","title":"Specification-Guided Data Aggregation for Semantically Aware Imitation Learning","date":"2023-03-29","arxiv_id":"2303.17010","n_code_links":0,"syntology":null},{"paper":"/paper/model-based-reinforcement-learning-with-3","slug":"model-based-reinforcement-learning-with-3","title":"Model-Based Reinforcement Learning with Isolated Imaginations","date":"2023-03-27","arxiv_id":"2303.14889","n_code_links":1,"syntology":null},{"paper":null,"slug":"implicit-ray-transformers-for-multi-view","title":"Implicit Ray-Transformers for Multi-view Remote Sensing Image Segmentation","date":"2023-03-15","arxiv_id":"2303.08401","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-based-wavefront","title":"Reinforcement Learning-based Wavefront Sensorless Adaptive Optics Approaches for Satellite-to-Ground Laser Communication","date":"2023-03-13","arxiv_id":"2303.07516","n_code_links":0,"syntology":null},{"paper":"/paper/twin-contrastive-learning-with-noisy-labels","slug":"twin-contrastive-learning-with-noisy-labels","title":"Twin Contrastive Learning with Noisy Labels","date":"2023-03-13","arxiv_id":"2303.06930","n_code_links":1,"syntology":null},{"paper":null,"slug":"self-nerf-a-self-training-pipeline-for-few","title":"Self-NeRF: A Self-Training Pipeline for Few-Shot Neural Radiance Fields","date":"2023-03-10","arxiv_id":"2303.05775","n_code_links":0,"syntology":null},{"paper":"/paper/intriguing-property-of-gan-for-remote-sensing","slug":"intriguing-property-of-gan-for-remote-sensing","title":"Intriguing Property and Counterfactual Explanation of GAN for Remote Sensing Image Generation","date":"2023-03-09","arxiv_id":"2303.05240","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-strategy-oriented-bayesian-soft-actor","title":"A Strategy-Oriented Bayesian Soft Actor-Critic Model","date":"2023-03-07","arxiv_id":"2303.04193","n_code_links":0,"syntology":null},{"paper":null,"slug":"uncoupled-and-convergent-learning-in-two","title":"Uncoupled and Convergent Learning in Two-Player Zero-Sum Markov Games with Bandit Feedback","date":"2023-03-05","arxiv_id":"2303.02738","n_code_links":0,"syntology":null},{"paper":null,"slug":"double-a3c-deep-reinforcement-learning-on","title":"Double A3C: Deep Reinforcement Learning on OpenAI Gym Games","date":"2023-03-04","arxiv_id":"2303.02271","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-reinforcement-learning-approach-for-7","title":"A Reinforcement Learning Approach for Scheduling Problems With Improved Generalization Through Order Swapping","date":"2023-02-27","arxiv_id":"2302.13941","n_code_links":0,"syntology":null},{"paper":null,"slug":"seo-safety-aware-energy-optimization","title":"SEO: Safety-Aware Energy Optimization Framework for Multi-Sensor Neural Controllers at the Edge","date":"2023-02-24","arxiv_id":"2302.12493","n_code_links":0,"syntology":null},{"paper":"/paper/behavior-proximal-policy-optimization","slug":"behavior-proximal-policy-optimization","title":"Behavior Proximal Policy Optimization","date":"2023-02-22","arxiv_id":"2302.11312","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["dragon-zhuang/bppo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/dynamic-simplex-balancing-safety-and","slug":"dynamic-simplex-balancing-safety-and","title":"Dynamic Simplex: Balancing Safety and Performance in Autonomous Cyber Physical Systems","date":"2023-02-20","arxiv_id":"2302.09750","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["BaitingLuo/Dynamic_Simplex"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"towards-co-operative-congestion-mitigation","title":"Towards Co-operative Congestion Mitigation","date":"2023-02-17","arxiv_id":"2302.09140","n_code_links":0,"syntology":null},{"paper":null,"slug":"cooperative-perception-for-safe-control-of","title":"Cooperative Perception for Safe Control of Autonomous Vehicles under LiDAR Spoofing Attacks","date":"2023-02-14","arxiv_id":"2302.07341","n_code_links":0,"syntology":null},{"paper":null,"slug":"energyshield-provably-safe-offloading-of","title":"EnergyShield: Provably-Safe Offloading of Neural Network Controllers for Energy Efficiency","date":"2023-02-13","arxiv_id":"2302.06572","n_code_links":0,"syntology":null},{"paper":null,"slug":"shared-information-based-safe-and-efficient","title":"Shared Information-Based Safe And Efficient Behavior Planning For Connected Autonomous Vehicles","date":"2023-02-08","arxiv_id":"2302.04321","n_code_links":0,"syntology":null},{"paper":"/paper/sample-dropout-a-simple-yet-effective","slug":"sample-dropout-a-simple-yet-effective","title":"Sample Dropout: A Simple yet Effective Variance Reduction Technique in Deep Policy Optimization","date":"2023-02-05","arxiv_id":"2302.02299","n_code_links":1,"syntology":{"ran":6,"of":12,"n_ran_checked":5,"n_instrument":1,"unverified":6,"pointer_only":12,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","official":{"repos":["linzichuan/sdpo"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/steps-joint-self-supervised-nighttime-image","slug":"steps-joint-self-supervised-nighttime-image","title":"STEPS: Joint Self-supervised Nighttime Image Enhancement and Depth Estimation","date":"2023-02-02","arxiv_id":"2302.01334","n_code_links":1,"syntology":null},{"paper":null,"slug":"bridging-physics-informed-neural-networks","title":"Bridging Physics-Informed Neural Networks with Reinforcement Learning: Hamilton-Jacobi-Bellman Proximal Policy Optimization (HJBPPO)","date":"2023-02-01","arxiv_id":"2302.00237","n_code_links":0,"syntology":null},{"paper":"/paper/learning-fast-and-slow-a-goal-directed-memory","slug":"learning-fast-and-slow-a-goal-directed-memory","title":"Learning, Fast and Slow: A Goal-Directed Memory-Based Approach for Dynamic Environments","date":"2023-01-31","arxiv_id":"2301.13758","n_code_links":1,"syntology":null},{"paper":"/paper/a-novel-framework-for-policy-mirror-descent","slug":"a-novel-framework-for-policy-mirror-descent","title":"A Novel Framework for Policy Mirror Descent with General Parameterization and Linear Convergence","date":"2023-01-30","arxiv_id":"2301.13139","n_code_links":1,"syntology":null},{"paper":null,"slug":"fast-computation-of-optimal-transport-via","title":"Fast Computation of Optimal Transport via Entropy-Regularized Extragradient Methods","date":"2023-01-30","arxiv_id":"2301.13006","n_code_links":0,"syntology":null},{"paper":null,"slug":"incorporating-recurrent-reinforcement","title":"Incorporating Recurrent Reinforcement Learning into Model Predictive Control for Adaptive Control in Autonomous Driving","date":"2023-01-30","arxiv_id":"2301.13313","n_code_links":0,"syntology":null},{"paper":null,"slug":"softtreemax-exponential-variance-reduction-in","title":"SoftTreeMax: Exponential Variance Reduction in Policy Gradient via Tree Search","date":"2023-01-30","arxiv_id":"2301.13236","n_code_links":0,"syntology":null},{"paper":"/paper/joint-action-loss-for-proximal-policy","slug":"joint-action-loss-for-proximal-policy","title":"Joint action loss for proximal policy optimization","date":"2023-01-26","arxiv_id":"2301.10919","n_code_links":1,"syntology":null},{"paper":null,"slug":"pdvn-a-patch-based-dual-view-network-for-face","title":"PDVN: A Patch-based Dual-view Network for Face Liveness Detection using Light Field Focal Stack","date":"2023-01-17","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/asynchronous-multi-agent-reinforcement","slug":"asynchronous-multi-agent-reinforcement","title":"Asynchronous Multi-Agent Reinforcement Learning for Efficient Real-Time Multi-Robot Cooperative Exploration","date":"2023-01-09","arxiv_id":"2301.03398","n_code_links":2,"syntology":null},{"paper":null,"slug":"tuning-path-tracking-controllers-for","title":"Tuning Path Tracking Controllers for Autonomous Cars Using Reinforcement Learning","date":"2023-01-09","arxiv_id":"2301.03363","n_code_links":0,"syntology":null},{"paper":null,"slug":"e-inu-simulating-a-quadruped-robot-with","title":"e-Inu: Simulating A Quadruped Robot With Emotional Sentience","date":"2023-01-03","arxiv_id":"2301.00964","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-for-asset-1","title":"Deep Reinforcement Learning for Asset Allocation: Reward Clipping","date":"2023-01-02","arxiv_id":"2301.05300","n_code_links":0,"syntology":null},{"paper":null,"slug":"simoun-synergizing-interactive-motion","title":"Simoun: Synergizing Interactive Motion-appearance Understanding for Vision-based Reinforcement Learning","date":"2023-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"stabilizing-visual-reinforcement-learning-via","title":"Stabilizing Visual Reinforcement Learning via Asymmetric Interactive Cooperation","date":"2023-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-automating-codenames-spymasters-with","title":"Towards automating Codenames spymasters with deep reinforcement learning","date":"2022-12-28","arxiv_id":"2212.14104","n_code_links":0,"syntology":null},{"paper":null,"slug":"hierarchical-deep-reinforcement-learning-for","title":"Hierarchical Deep Reinforcement Learning for Age-of-Information Minimization in IRS-aided and Wireless-powered Wireless Networks","date":"2022-12-27","arxiv_id":"2212.13390","n_code_links":0,"syntology":null},{"paper":"/paper/learning-generalizable-representations-for-1","slug":"learning-generalizable-representations-for-1","title":"Learning Generalizable Representations for Reinforcement Learning via Adaptive Meta-learner of Behavioral Similarities","date":"2022-12-26","arxiv_id":"2212.13088","n_code_links":1,"syntology":null},{"paper":null,"slug":"alignment-entropy-regularization","title":"Alignment Entropy Regularization","date":"2022-12-22","arxiv_id":"2212.12442","n_code_links":0,"syntology":null},{"paper":"/paper/lifelong-reinforcement-learning-with","slug":"lifelong-reinforcement-learning-with","title":"Lifelong Reinforcement Learning with Modulating Masks","date":"2022-12-21","arxiv_id":"2212.11110","n_code_links":4,"syntology":null},{"paper":null,"slug":"pre-trained-image-encoder-for-generalizable","title":"Pre-Trained Image Encoder for Generalizable Visual Reinforcement Learning","date":"2022-12-17","arxiv_id":"2212.08860","n_code_links":0,"syntology":null},{"paper":null,"slug":"distribution-aware-goal-prediction-and","title":"Distribution-aware Goal Prediction and Conformant Model-based Planning for Safe Autonomous Driving","date":"2022-12-16","arxiv_id":"2212.08729","n_code_links":0,"syntology":null},{"paper":"/paper/learning-for-vehicle-to-vehicle-cooperative","slug":"learning-for-vehicle-to-vehicle-cooperative","title":"Learning for Vehicle-to-Vehicle Cooperative Perception under Lossy Communication","date":"2022-12-16","arxiv_id":"2212.08273","n_code_links":1,"syntology":null},{"paper":null,"slug":"multi-agent-reinforcement-learning-with-4","title":"Multi-Agent Reinforcement Learning with Shared Resources for Inventory Management","date":"2022-12-15","arxiv_id":"2212.07684","n_code_links":0,"syntology":null},{"paper":"/paper/robust-policy-optimization-in-deep","slug":"robust-policy-optimization-in-deep","title":"Robust Policy Optimization in Deep Reinforcement Learning","date":"2022-12-14","arxiv_id":"2212.07536","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["vwxyzjn/cleanrl"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"paper":null,"slug":"ppo-ue-proximal-policy-optimization-via","title":"PPO-UE: Proximal Policy Optimization via Uncertainty-Aware Exploration","date":"2022-12-13","arxiv_id":"2212.06343","n_code_links":0,"syntology":null},{"paper":null,"slug":"decentralized-cooperative-perception-for","title":"Decentralized cooperative perception for autonomous vehicles: Learning to value the unknown","date":"2022-12-12","arxiv_id":"2301.01250","n_code_links":0,"syntology":null},{"paper":"/paper/solving-the-side-chain-packing-arrangement-of","slug":"solving-the-side-chain-packing-arrangement-of","title":"Reinforcement Learning for Molecular Dynamics Optimization: A Stochastic Pontryagin Maximum Principle Approach","date":"2022-12-06","arxiv_id":"2212.03320","n_code_links":1,"syntology":null},{"paper":null,"slug":"resilience-evaluation-of-entropy-regularized","title":"Resilience Evaluation of Entropy Regularized Logistic Networks with Probabilistic Cost","date":"2022-12-05","arxiv_id":"2212.02060","n_code_links":0,"syntology":null},{"paper":null,"slug":"safe-reinforcement-learning-with","title":"Safe Reinforcement Learning with Probabilistic Control Barrier Functions for Ramp Merging","date":"2022-12-01","arxiv_id":"2212.00618","n_code_links":0,"syntology":null},{"paper":null,"slug":"handling-missing-data-via-max-entropy","title":"Handling Missing Data via Max-Entropy Regularized Graph Autoencoder","date":"2022-11-30","arxiv_id":"2211.16771","n_code_links":0,"syntology":null},{"paper":"/paper/analyzing-infrastructure-lidar-placement-with","slug":"analyzing-infrastructure-lidar-placement-with","title":"Analyzing Infrastructure LiDAR Placement with Realistic LiDAR Simulation Library","date":"2022-11-29","arxiv_id":"2211.15975","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["pjlab-adg/lidarsimlib-and-placement-evaluation","pjlab-adg/pcsim"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"combined-peak-reduction-and-self-consumption","title":"Combined Peak Reduction and Self-Consumption Using Proximal Policy Optimization","date":"2022-11-27","arxiv_id":"2211.14831","n_code_links":0,"syntology":null},{"paper":"/paper/homology-constrained-vector-quantization","slug":"homology-constrained-vector-quantization","title":"Homology-constrained vector quantization entropy regularizer","date":"2022-11-25","arxiv_id":"2211.14363","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["st4cks1defl0w/hcvq"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"mutual-information-learned-regressor-an","title":"Mutual Information Learned Regressor: an Information-theoretic Viewpoint of Training Regression Systems","date":"2022-11-23","arxiv_id":"2211.12685","n_code_links":0,"syntology":null},{"paper":"/paper/multi-task-learning-for-camera-calibration","slug":"multi-task-learning-for-camera-calibration","title":"Multi-task Learning for Camera Calibration","date":"2022-11-22","arxiv_id":"2211.12432","n_code_links":2,"syntology":null},{"paper":null,"slug":"rationale-aware-autonomous-driving-policy","title":"Rationale-aware Autonomous Driving Policy utilizing Safety Force Field implemented on CARLA Simulator","date":"2022-11-18","arxiv_id":"2211.10237","n_code_links":0,"syntology":null},{"paper":"/paper/dynamic-conditional-imitation-learning-for-1","slug":"dynamic-conditional-imitation-learning-for-1","title":"Dynamic Conditional Imitation Learning for Autonomous Driving","date":"2022-11-17","arxiv_id":"2211.11579","n_code_links":1,"syntology":null}],"record_sha256":"ab3ca6d159d6a9434938db57f6e4c411efd686c55c5480f0ec69b2b738f093bf","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}