{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/entropy-regularization/papers/10","list_of":"/method/entropy-regularization","method":"Entropy Regularization","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":10,"pages_in_order":12,"rows_per_page":100,"rows":[901,1000],"of":1128,"counts":{"archive_papers_tagged":1128,"with_a_code_link":451,"where_syntology_ran_a_sample":156,"not_listed_spam_title":0,"listed":1128,"listed_where_code_ran":156,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":129,"every_run_a_failure_of_syntologys_instrument":27,"listed_with_a_run_with_no_instrument_failure":129,"listed_every_run_a_failure_of_syntologys_instrument":27,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/entropy-regularization","prev":"/method/entropy-regularization/papers/9","next":"/method/entropy-regularization/papers/11","papers":[{"paper":null,"slug":"manual-label-free-3d-detection-via-an-open","title":"Manual-Label Free 3D Detection via An Open-Source Simulator","date":"2020-11-16","arxiv_id":"2011.07784","n_code_links":0,"syntology":null},{"paper":"/paper/tonic-a-deep-reinforcement-learning-library","slug":"tonic-a-deep-reinforcement-learning-library","title":"Tonic: A Deep Reinforcement Learning Library for Fast Prototyping and Benchmarking","date":"2020-11-15","arxiv_id":"2011.07537","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":0,"n_instrument":3,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["fabiopardo/tonic"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/query-based-targeted-action-space-adversarial","slug":"query-based-targeted-action-space-adversarial","title":"Query-based Targeted Action-Space Adversarial Policies on Deep Reinforcement Learning Agents","date":"2020-11-13","arxiv_id":"2011.07114","n_code_links":1,"syntology":null},{"paper":null,"slug":"proximal-policy-optimization-via-enhanced","title":"Proximal Policy Optimization via Enhanced Exploration Efficiency","date":"2020-11-11","arxiv_id":"2011.05525","n_code_links":0,"syntology":null},{"paper":"/paper/trajectory-planning-for-autonomous-vehicles","slug":"trajectory-planning-for-autonomous-vehicles","title":"Trajectory Planning for Autonomous Vehicles Using Hierarchical Reinforcement Learning","date":"2020-11-09","arxiv_id":"2011.04752","n_code_links":1,"syntology":null},{"paper":"/paper/multimodal-trajectory-prediction-via","slug":"multimodal-trajectory-prediction-via","title":"Multimodal Trajectory Prediction via Topological Invariance for Navigation at Uncontrolled Intersections","date":"2020-11-08","arxiv_id":"2011.03894","n_code_links":1,"syntology":null},{"paper":"/paper/drafting-in-collectible-card-games-via","slug":"drafting-in-collectible-card-games-via","title":"Drafting in Collectible Card Games via Reinforcement Learning","date":"2020-11-07","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/guided-dialogue-policy-learning-without","slug":"guided-dialogue-policy-learning-without","title":"Guided Dialogue Policy Learning without Adversarial Learning in the Loop","date":"2020-11-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"pilot-efficient-planning-by-imitation","title":"PILOT: Efficient Planning by Imitation Learning and Optimisation for Safe Autonomous Driving","date":"2020-11-01","arxiv_id":"2011.00509","n_code_links":0,"syntology":null},{"paper":"/paper/a-software-architecture-for-autonomous","slug":"a-software-architecture-for-autonomous","title":"A Software Architecture for Autonomous Vehicles: Team LRM-B Entry in the First CARLA Autonomous Driving Challenge","date":"2020-10-23","arxiv_id":"2010.12598","n_code_links":0,"syntology":null},{"paper":null,"slug":"proximal-policy-gradient-ppo-with-policy","title":"Proximal Policy Gradient: PPO with Policy Gradient","date":"2020-10-20","arxiv_id":"2010.09933","n_code_links":0,"syntology":null},{"paper":"/paper/finding-physical-adversarial-examples-for-1","slug":"finding-physical-adversarial-examples-for-1","title":"Finding Physical Adversarial Examples for Autonomous Driving with Fast and Differentiable Image Compositing","date":"2020-10-17","arxiv_id":"2010.08844","n_code_links":1,"syntology":null},{"paper":"/paper/learning-monocular-dense-depth-from-events","slug":"learning-monocular-dense-depth-from-events","title":"Learning Monocular Dense Depth from Events","date":"2020-10-16","arxiv_id":"2010.08350","n_code_links":1,"syntology":null},{"paper":null,"slug":"recurrent-distributed-reinforcement-learning","title":"A Learning Approach to Robot-Agnostic Force-Guided High Precision Assembly","date":"2020-10-15","arxiv_id":"2010.08052","n_code_links":0,"syntology":null},{"paper":"/paper/unsupervised-learning-of-depth-and-ego-motion-3","slug":"unsupervised-learning-of-depth-and-ego-motion-3","title":"Unsupervised Learning of Depth and Ego-Motion from Cylindrical Panoramic Video with Applications for Virtual Reality","date":"2020-10-14","arxiv_id":"2010.07704","n_code_links":1,"syntology":null},{"paper":null,"slug":"lm-reloc-levenberg-marquardt-based-direct","title":"LM-Reloc: Levenberg-Marquardt Based Direct Visual Relocalization","date":"2020-10-13","arxiv_id":"2010.06323","n_code_links":0,"syntology":null},{"paper":"/paper/discrete-latent-space-world-models-for","slug":"discrete-latent-space-world-models-for","title":"Smaller World Models for Reinforcement Learning","date":"2020-10-12","arxiv_id":"2010.05767","n_code_links":0,"syntology":null},{"paper":"/paper/automated-concatenation-of-embeddings-for-1","slug":"automated-concatenation-of-embeddings-for-1","title":"Automated Concatenation of Embeddings for Structured Prediction","date":"2020-10-10","arxiv_id":"2010.05006","n_code_links":2,"syntology":null},{"paper":"/paper/no-mcmc-for-me-amortized-sampling-for-fast-1","slug":"no-mcmc-for-me-amortized-sampling-for-fast-1","title":"No MCMC for me: Amortized sampling for fast and stable training of energy-based models","date":"2020-10-08","arxiv_id":"2010.04230","n_code_links":1,"syntology":null},{"paper":null,"slug":"proximal-policy-optimization-with-relative","title":"Proximal Policy Optimization with Relative Pearson Divergence","date":"2020-10-07","arxiv_id":"2010.03290","n_code_links":0,"syntology":null},{"paper":"/paper/neural-mask-generator-learning-to-generate","slug":"neural-mask-generator-learning-to-generate","title":"Neural Mask Generator: Learning to Generate Adaptive Word Maskings for Language Model Adaptation","date":"2020-10-06","arxiv_id":"2010.02705","n_code_links":1,"syntology":null},{"paper":null,"slug":"entropy-regularization-for-mean-field-games","title":"Entropy Regularization for Mean Field Games with Learning","date":"2020-09-30","arxiv_id":"2010.00145","n_code_links":0,"syntology":null},{"paper":"/paper/revisiting-design-choices-in-proximal-policy","slug":"revisiting-design-choices-in-proximal-policy","title":"Revisiting Design Choices in Proximal Policy Optimization","date":"2020-09-23","arxiv_id":"2009.10897","n_code_links":1,"syntology":{"ran":5,"of":11,"n_ran_checked":4,"n_instrument":1,"unverified":6,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","official":{"repos":["chloechsu/revisiting-ppo"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"regularizing-attention-networks-for-anomaly","title":"Regularizing Attention Networks for Anomaly Detection in Visual Question Answering","date":"2020-09-21","arxiv_id":"2009.10054","n_code_links":0,"syntology":null},{"paper":"/paper/phasic-policy-gradient","slug":"phasic-policy-gradient","title":"Phasic Policy Gradient","date":"2020-09-09","arxiv_id":"2009.04416","n_code_links":3,"syntology":{"ran":7,"of":13,"n_ran_checked":7,"n_instrument":0,"unverified":6,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","official":{"repos":["openai/phasic-policy-gradient"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"data-driven-transferred-energy-management","title":"Data-Driven Transferred Energy Management Strategy for Hybrid Electric Vehicles via Deep Reinforcement Learning","date":"2020-09-07","arxiv_id":"2009.03289","n_code_links":0,"syntology":null},{"paper":"/paper/drle-decentralized-reinforcement-learning-at","slug":"drle-decentralized-reinforcement-learning-at","title":"DRLE: Decentralized Reinforcement Learning at the Edge for Traffic Light Control in the IoV","date":"2020-09-03","arxiv_id":"2009.01502","n_code_links":1,"syntology":null},{"paper":"/paper/dynamic-scheduling-for-stochastic-edge-cloud","slug":"dynamic-scheduling-for-stochastic-edge-cloud","title":"Dynamic Scheduling for Stochastic Edge-Cloud Computing Environments using A3C learning and Residual Recurrent Neural Networks","date":"2020-09-01","arxiv_id":"2009.02186","n_code_links":1,"syntology":null},{"paper":null,"slug":"driving-through-ghosts-behavioral-cloning","title":"Driving Through Ghosts: Behavioral Cloning with False Positives","date":"2020-08-29","arxiv_id":"2008.12969","n_code_links":0,"syntology":null},{"paper":"/paper/on-the-model-based-stochastic-value-gradient","slug":"on-the-model-based-stochastic-value-gradient","title":"On the model-based stochastic value gradient for continuous reinforcement learning","date":"2020-08-28","arxiv_id":"2008.12775","n_code_links":1,"syntology":null},{"paper":"/paper/domain-adaptation-through-task-distillation","slug":"domain-adaptation-through-task-distillation","title":"Domain Adaptation Through Task Distillation","date":"2020-08-27","arxiv_id":"2008.11911","n_code_links":1,"syntology":null},{"paper":"/paper/query-focused-multi-document-summarisation-of-1","slug":"query-focused-multi-document-summarisation-of-1","title":"Query Focused Multi-document Summarisation of Biomedical Texts: Macquarie Universiy and the Australian National University at BioASQ8b","date":"2020-08-27","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/cross-regional-oil-palm-tree-counting-and","slug":"cross-regional-oil-palm-tree-counting-and","title":"Cross-regional oil palm tree counting and detection via multi-level attention domain adaptation network","date":"2020-08-26","arxiv_id":"2008.11505","n_code_links":1,"syntology":null},{"paper":"/paper/towards-closing-the-sim-to-real-gap-in","slug":"towards-closing-the-sim-to-real-gap-in","title":"Towards Closing the Sim-to-Real Gap in Collaborative Multi-Robot Deep Reinforcement Learning","date":"2020-08-18","arxiv_id":"2008.07875","n_code_links":1,"syntology":null},{"paper":null,"slug":"reinforced-wasserstein-training-for-severity","title":"Reinforced Wasserstein Training for Severity-Aware Semantic Segmentation in Autonomous Driving","date":"2020-08-11","arxiv_id":"2008.04751","n_code_links":0,"syntology":null},{"paper":null,"slug":"physical-adversarial-attack-on-vehicle","title":"Physical Adversarial Attack on Vehicle Detector in the Carla Simulator","date":"2020-07-31","arxiv_id":"2007.16118","n_code_links":0,"syntology":null},{"paper":"/paper/queueing-network-controls-via-deep","slug":"queueing-network-controls-via-deep","title":"Queueing Network Controls via Deep Reinforcement Learning","date":"2020-07-31","arxiv_id":"2008.01644","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"lagrangian-duality-in-reinforcement-learning","title":"Lagrangian Duality in Reinforcement Learning","date":"2020-07-20","arxiv_id":"2007.09998","n_code_links":0,"syntology":null},{"paper":null,"slug":"fast-global-convergence-of-natural-policy","title":"Fast Global Convergence of Natural Policy Gradient Methods with Entropy Regularization","date":"2020-07-13","arxiv_id":"2007.06558","n_code_links":0,"syntology":null},{"paper":null,"slug":"maximum-entropy-regularization-and-chinese","title":"Maximum Entropy Regularization and Chinese Text Recognition","date":"2020-07-09","arxiv_id":"2007.04651","n_code_links":0,"syntology":null},{"paper":"/paper/learning-implicit-credit-assignment-for-multi","slug":"learning-implicit-credit-assignment-for-multi","title":"Learning Implicit Credit Assignment for Cooperative Multi-Agent Reinforcement Learning","date":"2020-07-06","arxiv_id":"2007.02529","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["mzho7212/LICA"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"directional-primitives-for-uncertainty-aware","title":"Directional Primitives for Uncertainty-Aware Motion Estimation in Urban Environments","date":"2020-07-01","arxiv_id":"2007.00161","n_code_links":0,"syntology":null},{"paper":"/paper/sample-factory-egocentric-3d-control-from","slug":"sample-factory-egocentric-3d-control-from","title":"Sample Factory: Egocentric 3D Control from Pixels at 100000 FPS with Asynchronous Reinforcement Learning","date":"2020-06-21","arxiv_id":"2006.11751","n_code_links":4,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["alex-petrenko/sample-factory"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"an-operator-view-of-policy-gradient-methods","title":"An operator view of policy gradient methods","date":"2020-06-19","arxiv_id":"2006.11266","n_code_links":0,"syntology":null},{"paper":null,"slug":"generalization-of-agent-behavior-through","title":"Generalization of Agent Behavior through Explicit Representation of Context","date":"2020-06-18","arxiv_id":"2006.11305","n_code_links":0,"syntology":null},{"paper":"/paper/fine-tuning-darts-for-image-classification","slug":"fine-tuning-darts-for-image-classification","title":"Fine-Tuning DARTS for Image Classification","date":"2020-06-16","arxiv_id":"2006.09042","n_code_links":0,"syntology":null},{"paper":null,"slug":"shieldnn-a-provably-safe-nn-filter-for-unsafe","title":"ShieldNN: A Provably Safe NN Filter for Unsafe NN Controllers","date":"2020-06-16","arxiv_id":"2006.09564","n_code_links":0,"syntology":null},{"paper":"/paper/optimistic-distributionally-robust-policy","slug":"optimistic-distributionally-robust-policy","title":"Optimistic Distributionally Robust Policy Optimization","date":"2020-06-14","arxiv_id":"2006.07815","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["kadysongbb/dr-trpo"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/bonsai-net-one-shot-neural-architecture","slug":"bonsai-net-one-shot-neural-architecture","title":"Bonsai-Net: One-Shot Neural Architecture Search via Differentiable Pruners","date":"2020-06-12","arxiv_id":"2006.09264","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["RobGeada/bonsai-net-lite"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"exploration-by-maximizing-renyi-entropy-for","title":"Exploration by Maximizing Rényi Entropy for Reward-Free RL Framework","date":"2020-06-11","arxiv_id":"2006.06193","n_code_links":0,"syntology":null},{"paper":"/paper/rethinking-pre-training-and-self-training","slug":"rethinking-pre-training-and-self-training","title":"Rethinking Pre-training and Self-training","date":"2020-06-11","arxiv_id":"2006.06882","n_code_links":2,"syntology":null},{"paper":null,"slug":"learning-navigation-costs-from-demonstration-1","title":"Learning Navigation Costs from Demonstration with Semantic Observations","date":"2020-06-09","arxiv_id":"2006.05043","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-comparison-of-self-play-algorithms-under-a","title":"A Comparison of Self-Play Algorithms Under a Generalized Framework","date":"2020-06-08","arxiv_id":"2006.04471","n_code_links":0,"syntology":null},{"paper":null,"slug":"fast-synthetic-lidar-rendering-via-spherical","title":"Fast Synthetic LiDAR Rendering via Spherical UV Unwrapping of Equirectangular Z-Buffer Images","date":"2020-06-08","arxiv_id":"2006.04345","n_code_links":0,"syntology":null},{"paper":null,"slug":"explaining-autonomous-driving-by-learning-end","title":"Explaining Autonomous Driving by Learning End-to-End Visual Attention","date":"2020-06-05","arxiv_id":"2006.03347","n_code_links":0,"syntology":null},{"paper":"/paper/optimization-and-passive-flow-control-using","slug":"optimization-and-passive-flow-control-using","title":"Single-step deep reinforcement learning for open-loop control of laminar and turbulent flows","date":"2020-06-04","arxiv_id":"2006.02979","n_code_links":1,"syntology":null},{"paper":"/paper/diversity-actor-critic-sample-aware-entropy","slug":"diversity-actor-critic-sample-aware-entropy","title":"Diversity Actor-Critic: Sample-Aware Entropy Regularization for Sample-Efficient Exploration","date":"2020-06-02","arxiv_id":"2006.01419","n_code_links":1,"syntology":null},{"paper":"/paper/exploring-data-aggregation-in-policy-learning","slug":"exploring-data-aggregation-in-policy-learning","title":"Exploring Data Aggregation in Policy Learning for Vision-Based Urban Autonomous Driving","date":"2020-06-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-situational-driving","title":"Learning Situational Driving","date":"2020-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"severity-aware-semantic-segmentation-with","title":"Severity-Aware Semantic Segmentation With Reinforced Wasserstein Training","date":"2020-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/fast-risk-assessment-for-autonomous-vehicles","slug":"fast-risk-assessment-for-autonomous-vehicles","title":"Fast Risk Assessment for Autonomous Vehicles Using Learned Models of Agent Futures","date":"2020-05-27","arxiv_id":"2005.13458","n_code_links":1,"syntology":null},{"paper":null,"slug":"dynamic-value-estimation-for-single-task","title":"Dynamic Value Estimation for Single-Task Multi-Scene Reinforcement Learning","date":"2020-05-25","arxiv_id":"2005.12254","n_code_links":0,"syntology":null},{"paper":"/paper/implementation-matters-in-deep-policy","slug":"implementation-matters-in-deep-policy","title":"Implementation Matters in Deep Policy Gradients: A Case Study on PPO and TRPO","date":"2020-05-25","arxiv_id":"2005.12729","n_code_links":3,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["MadryLab/implementation-matters"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/mirror-descent-policy-optimization","slug":"mirror-descent-policy-optimization","title":"Mirror Descent Policy Optimization","date":"2020-05-20","arxiv_id":"2005.09814","n_code_links":1,"syntology":null},{"paper":null,"slug":"on-the-global-convergence-rates-of-softmax","title":"On the Global Convergence Rates of Softmax Policy Gradient Methods","date":"2020-05-13","arxiv_id":"2005.06392","n_code_links":0,"syntology":null},{"paper":"/paper/generalized-state-dependent-exploration-for","slug":"generalized-state-dependent-exploration-for","title":"Smooth Exploration for Robotic Reinforcement Learning","date":"2020-05-12","arxiv_id":"2005.05719","n_code_links":4,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["DLR-RM/stable-baselines3"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/learning-hierarchical-behavior-and-motion","slug":"learning-hierarchical-behavior-and-motion","title":"Learning hierarchical behavior and motion planning for autonomous driving","date":"2020-05-08","arxiv_id":"2005.03863","n_code_links":1,"syntology":null},{"paper":null,"slug":"generalized-entropy-regularization-or-there-s","title":"Generalized Entropy Regularization or: There's Nothing Special about Label Smoothing","date":"2020-05-02","arxiv_id":"2005.00820","n_code_links":0,"syntology":null},{"paper":null,"slug":"model-based-reinforcement-learning-for","title":"Model-based reinforcement learning for biological sequence design","date":"2020-05-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/look-at-the-first-sentence-position-bias-in","slug":"look-at-the-first-sentence-position-bias-in","title":"Look at the First Sentence: Position Bias in Question Answering","date":"2020-04-30","arxiv_id":"2004.14602","n_code_links":1,"syntology":null},{"paper":"/paper/reinforcement-learning-with-augmented-data","slug":"reinforcement-learning-with-augmented-data","title":"Reinforcement Learning with Augmented Data","date":"2020-04-30","arxiv_id":"2004.14990","n_code_links":2,"syntology":{"ran":21,"of":24,"n_ran_checked":9,"n_instrument":12,"unverified":3,"pointer_only":21,"phrase":"21 ran (of which 8 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 12 where Syntology's instrument failed) · 3 unverified","official":null}},{"paper":"/paper/per-step-reward-a-new-perspective-for-risk","slug":"per-step-reward-a-new-perspective-for-risk","title":"Mean-Variance Policy Iteration for Risk-Averse Reinforcement Learning","date":"2020-04-22","arxiv_id":"2004.10888","n_code_links":1,"syntology":null},{"paper":null,"slug":"parkpredict-motion-and-intent-prediction-of","title":"ParkPredict: Motion and Intent Prediction of Vehicles in Parking Lots","date":"2020-04-21","arxiv_id":"2004.10293","n_code_links":0,"syntology":null},{"paper":"/paper/solving-the-scalarization-issues-of-advantage","slug":"solving-the-scalarization-issues-of-advantage","title":"Solving the scalarization issues of Advantage-based Reinforcement Learning Algorithms","date":"2020-04-08","arxiv_id":"2004.04120","n_code_links":1,"syntology":null},{"paper":"/paper/guided-dialog-policy-learning-without","slug":"guided-dialog-policy-learning-without","title":"Guided Dialog Policy Learning without Adversarial Learning in the Loop","date":"2020-04-07","arxiv_id":"2004.03267","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["cszmli/dp-without-adv"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"paper":"/paper/evolving-normalization-activation-layers","slug":"evolving-normalization-activation-layers","title":"Evolving Normalization-Activation Layers","date":"2020-04-06","arxiv_id":"2004.02967","n_code_links":8,"syntology":{"ran":3,"of":5,"n_ran_checked":0,"n_instrument":3,"unverified":2,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":null,"slug":"leverage-the-average-an-analysis-of","title":"Leverage the Average: an Analysis of KL Regularization in RL","date":"2020-03-31","arxiv_id":"2003.14089","n_code_links":0,"syntology":null},{"paper":"/paper/mtl-nas-task-agnostic-neural-architecture","slug":"mtl-nas-task-agnostic-neural-architecture","title":"MTL-NAS: Task-Agnostic Neural Architecture Search towards General-Purpose Multi-Task Learning","date":"2020-03-31","arxiv_id":"2003.14058","n_code_links":1,"syntology":{"ran":9,"of":13,"n_ran_checked":7,"n_instrument":2,"unverified":4,"pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","official":{"repos":["bhpfelix/MTLNAS"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/obstacle-avoidance-and-navigation-utilizing","slug":"obstacle-avoidance-and-navigation-utilizing","title":"Obstacle Avoidance and Navigation Utilizing Reinforcement Learning with Reward Shaping","date":"2020-03-28","arxiv_id":"2003.12863","n_code_links":1,"syntology":null},{"paper":"/paper/towards-safer-self-driving-through-great-pain","slug":"towards-safer-self-driving-through-great-pain","title":"Towards Safer Self-Driving Through Great PAIN (Physically Adversarial Intelligent Networks)","date":"2020-03-24","arxiv_id":"2003.10662","n_code_links":1,"syntology":null},{"paper":"/paper/robust-deep-reinforcement-learning-against","slug":"robust-deep-reinforcement-learning-against","title":"Robust Deep Reinforcement Learning against Adversarial Perturbations on State Observations","date":"2020-03-19","arxiv_id":"2003.08938","n_code_links":4,"syntology":null},{"paper":"/paper/particle-based-adaptive-discretization-for","slug":"particle-based-adaptive-discretization-for","title":"PFPN: Continuous Control of Physically Simulated Characters using Particle Filtering Policy Network","date":"2020-03-16","arxiv_id":"2003.06959","n_code_links":1,"syntology":null},{"paper":"/paper/explore-and-exploit-with-heterotic-line","slug":"explore-and-exploit-with-heterotic-line","title":"Explore and Exploit with Heterotic Line Bundle Models","date":"2020-03-10","arxiv_id":"2003.04817","n_code_links":1,"syntology":null},{"paper":"/paper/fast-online-adaptation-in-robotics-through","slug":"fast-online-adaptation-in-robotics-through","title":"Fast Online Adaptation in Robotics through Meta-Learning Embeddings of Simulated Priors","date":"2020-03-10","arxiv_id":"2003.04663","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-machine-learning-environment-for-evaluating","title":"A machine learning environment for evaluating autonomous driving software","date":"2020-03-07","arxiv_id":"2003.03576","n_code_links":0,"syntology":null},{"paper":null,"slug":"asynchronous-policy-evaluation-in-distributed","title":"Fully Asynchronous Policy Evaluation in Distributed Reinforcement Learning over Networks","date":"2020-03-01","arxiv_id":"2003.00433","n_code_links":0,"syntology":null},{"paper":null,"slug":"self-tuning-deep-reinforcement-learning","title":"A Self-Tuning Actor-Critic Algorithm","date":"2020-02-28","arxiv_id":"2002.12928","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-visual-communication-map-for-multi-agent","title":"A Visual Communication Map for Multi-Agent Deep Reinforcement Learning","date":"2020-02-27","arxiv_id":"2002.11882","n_code_links":0,"syntology":null},{"paper":"/paper/generalized-product-quantization-network-for","slug":"generalized-product-quantization-network-for","title":"Generalized Product Quantization Network for Semi-supervised Image Retrieval","date":"2020-02-26","arxiv_id":"2002.11281","n_code_links":2,"syntology":null},{"paper":"/paper/reinforcement-learning-framework-for-deep","slug":"reinforcement-learning-framework-for-deep","title":"Reinforcement Learning Framework for Deep Brain Stimulation Study","date":"2020-02-22","arxiv_id":"2002.10948","n_code_links":1,"syntology":null},{"paper":"/paper/first-order-optimization-in-policy-space-for","slug":"first-order-optimization-in-policy-space-for","title":"First Order Constrained Optimization in Policy Space","date":"2020-02-16","arxiv_id":"2002.06506","n_code_links":2,"syntology":null},{"paper":"/paper/deep-rl-agent-for-a-real-time-action-strategy","slug":"deep-rl-agent-for-a-real-time-action-strategy","title":"Deep RL Agent for a Real-Time Action Strategy Game","date":"2020-02-15","arxiv_id":"2002.06290","n_code_links":1,"syntology":null},{"paper":null,"slug":"temporal-adaptive-hierarchical-reinforcement","title":"Temporal-adaptive Hierarchical Reinforcement Learning","date":"2020-02-06","arxiv_id":"2002.02080","n_code_links":0,"syntology":null},{"paper":null,"slug":"unsupervised-domain-adaptive-object-detection","title":"Unsupervised Domain Adaptive Object Detection using Forward-Backward Cyclic Adaptation","date":"2020-02-03","arxiv_id":"2002.00575","n_code_links":0,"syntology":null},{"paper":"/paper/integrating-deep-reinforcement-learning-with","slug":"integrating-deep-reinforcement-learning-with","title":"Integrating Deep Reinforcement Learning with Model-based Path Planners for Automated Driving","date":"2020-02-02","arxiv_id":"2002.00434","n_code_links":1,"syntology":{"ran":2,"of":5,"n_ran_checked":2,"n_instrument":0,"unverified":3,"pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["Ekim-Yurtsever/Hybrid-DeepRL-Automated-Driving"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"brain-metastasis-segmentation-network-trained","title":"Brain Metastasis Segmentation Network Trained with Robustness to Annotations with Multiple False Negatives","date":"2020-01-26","arxiv_id":"2001.09501","n_code_links":0,"syntology":null},{"paper":"/paper/interpretable-end-to-end-urban-autonomous","slug":"interpretable-end-to-end-urban-autonomous","title":"Interpretable End-to-end Urban Autonomous Driving with Latent Deep Reinforcement Learning","date":"2020-01-23","arxiv_id":"2001.08726","n_code_links":4,"syntology":{"ran":2,"of":6,"n_ran_checked":2,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["cjy1992/interp-e2e-driving"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/continuous-action-reinforcement-learning-for","slug":"continuous-action-reinforcement-learning-for","title":"Continuous-action Reinforcement Learning for Playing Racing Games: Comparing SPG to PPO","date":"2020-01-15","arxiv_id":"2001.05270","n_code_links":1,"syntology":null},{"paper":null,"slug":"intelligent-roundabout-insertion-using-deep","title":"Intelligent Roundabout Insertion using Deep Reinforcement Learning","date":"2020-01-03","arxiv_id":"2001.00786","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-representations-in-reinforcement-1","title":"Learning Representations in Reinforcement Learning: an Information Bottleneck Approach","date":"2020-01-01","arxiv_id":null,"n_code_links":0,"syntology":null}],"record_sha256":"617f3450fc5fe9863641c056f86da7d4a330b8c80197f17adbaae5103a7cd99b","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}