{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/entropy-regularization/papers/11","list_of":"/method/entropy-regularization","method":"Entropy Regularization","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":11,"pages_in_order":12,"rows_per_page":100,"rows":[1001,1100],"of":1128,"counts":{"archive_papers_tagged":1128,"with_a_code_link":451,"where_syntology_ran_a_sample":156,"not_listed_spam_title":0,"listed":1128,"listed_where_code_ran":156,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":129,"every_run_a_failure_of_syntologys_instrument":27,"listed_with_a_run_with_no_instrument_failure":129,"listed_every_run_a_failure_of_syntologys_instrument":27,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/entropy-regularization","prev":"/method/entropy-regularization/papers/10","next":"/method/entropy-regularization/papers/12","papers":[{"paper":null,"slug":"tpo-tree-search-policy-optimization-for","title":"TPO: TREE SEARCH POLICY OPTIMIZATION FOR CONTINUOUS ACTION SPACES","date":"2020-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/slm-lab-a-comprehensive-benchmark-and-modular-1","slug":"slm-lab-a-comprehensive-benchmark-and-modular-1","title":"SLM Lab: A Comprehensive Benchmark and Modular Software Framework for Reproducible Deep Reinforcement Learning","date":"2019-12-28","arxiv_id":"1912.12482","n_code_links":1,"syntology":{"ran":10,"of":13,"n_ran_checked":10,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["kengz/SLM-Lab"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/learning-by-cheating","slug":"learning-by-cheating","title":"Learning by Cheating","date":"2019-12-27","arxiv_id":"1912.12294","n_code_links":9,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["dotchen/LearningByCheating"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"mastering-complex-control-in-moba-games-with","title":"Mastering Complex Control in MOBA Games with Deep Reinforcement Learning","date":"2019-12-20","arxiv_id":"1912.09729","n_code_links":0,"syntology":null},{"paper":null,"slug":"soft-q-network","title":"Soft Q Network","date":"2019-12-20","arxiv_id":"1912.10891","n_code_links":0,"syntology":null},{"paper":null,"slug":"entropy-regularization-with-discounted-future","title":"Entropy Regularization with Discounted Future State Distribution in Policy Gradient Methods","date":"2019-12-11","arxiv_id":"1912.05104","n_code_links":0,"syntology":null},{"paper":null,"slug":"marginalized-state-distribution-entropy","title":"Marginalized State Distribution Entropy Regularization in Policy Optimization","date":"2019-12-11","arxiv_id":"1912.05128","n_code_links":0,"syntology":null},{"paper":"/paper/spinenet-learning-scale-permuted-backbone-for","slug":"spinenet-learning-scale-permuted-backbone-for","title":"SpineNet: Learning Scale-Permuted Backbone for Recognition and Localization","date":"2019-12-10","arxiv_id":"1912.05027","n_code_links":13,"syntology":null},{"paper":null,"slug":"intelligent-coordination-among-multiple","title":"Intelligent Coordination among Multiple Traffic Intersections Using Multi-Agent Reinforcement Learning","date":"2019-12-09","arxiv_id":"1912.03851","n_code_links":0,"syntology":null},{"paper":"/paper/lates-latent-space-distillation-for-teacher","slug":"lates-latent-space-distillation-for-teacher","title":"SAM: Squeeze-and-Mimic Networks for Conditional Visual Driving Policy Learning","date":"2019-12-06","arxiv_id":"1912.02973","n_code_links":1,"syntology":null},{"paper":"/paper/mnasfpn-learning-latency-aware-pyramid","slug":"mnasfpn-learning-latency-aware-pyramid","title":"MnasFPN: Learning Latency-aware Pyramid Architecture for Object Detection on Mobile Devices","date":"2019-12-02","arxiv_id":"1912.01106","n_code_links":2,"syntology":null},{"paper":null,"slug":"on-policy-reinforcement-learning-with-entropy","title":"Policy Optimization Reinforcement Learning with Entropy Regularization","date":"2019-12-02","arxiv_id":"1912.01557","n_code_links":0,"syntology":null},{"paper":null,"slug":"adversary-a3c-for-robust-reinforcement-1","title":"Adversary A3C for Robust Reinforcement Learning","date":"2019-12-01","arxiv_id":"1912.00330","n_code_links":0,"syntology":null},{"paper":"/paper/automated-curriculum-generation-for-policy","slug":"automated-curriculum-generation-for-policy","title":"Automated curriculum generation for Policy Gradients from Demonstrations","date":"2019-12-01","arxiv_id":"1912.00444","n_code_links":1,"syntology":null},{"paper":"/paper/learning-reward-machines-for-partially","slug":"learning-reward-machines-for-partially","title":"Learning Reward Machines for Partially Observable Reinforcement Learning","date":"2019-12-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"neural-trust-regionproximal-policy","title":"Neural Trust Region/Proximal Policy Optimization Attains Globally Optimal Policy","date":"2019-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"impact-importance-weighted-asynchronous-1","title":"IMPACT: Importance Weighted Asynchronous Architectures with Clipped Target Networks","date":"2019-11-30","arxiv_id":"1912.00167","n_code_links":0,"syntology":null},{"paper":"/paper/end-to-end-model-free-reinforcement-learning","slug":"end-to-end-model-free-reinforcement-learning","title":"End-to-End Model-Free Reinforcement Learning for Urban Driving using Implicit Affordances","date":"2019-11-25","arxiv_id":"1911.10868","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":4,"n_instrument":1,"unverified":2,"pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["valeoai/LearningByCheating"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"unsupervised-neural-sensor-models-for","title":"Unsupervised Neural Sensor Models for Synthetic LiDAR Data Augmentation","date":"2019-11-24","arxiv_id":"1911.10575","n_code_links":0,"syntology":null},{"paper":null,"slug":"accelerating-training-in-pommerman-with","title":"Accelerating Training in Pommerman with Imitation and Reinforcement Learning","date":"2019-11-12","arxiv_id":"1911.04947","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-representations-in-reinforcement","title":"Learning Representations in Reinforcement Learning:An Information Bottleneck Approach","date":"2019-11-12","arxiv_id":"1911.05695","n_code_links":0,"syntology":null},{"paper":null,"slug":"situated-gail-multitask-imitation-using-task","title":"Situated GAIL: Multitask imitation using task-conditioned adversarial inverse reinforcement learning","date":"2019-11-01","arxiv_id":"1911.00238","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-modal-semantic-segmentation-using","title":"Multi Modal Semantic Segmentation using Synthetic Data","date":"2019-10-30","arxiv_id":"1910.13676","n_code_links":0,"syntology":null},{"paper":"/paper/hrl4in-hierarchical-reinforcement-learning","slug":"hrl4in-hierarchical-reinforcement-learning","title":"HRL4IN: Hierarchical Reinforcement Learning for Interactive Navigation with Mobile Manipulators","date":"2019-10-24","arxiv_id":"1910.11432","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/regularization-matters-in-policy-optimization-1","slug":"regularization-matters-in-policy-optimization-1","title":"Regularization Matters in Policy Optimization","date":"2019-10-21","arxiv_id":"1910.09191","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["xuanlinli17/iclr2021_rlreg","xuanlinli17/po-rl-regularization"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"paper":null,"slug":"conditional-driving-from-natural-language","title":"Conditional Driving from Natural Language Instructions","date":"2019-10-16","arxiv_id":"1910.07615","n_code_links":0,"syntology":null},{"paper":"/paper/prescribed-generative-adversarial-networks","slug":"prescribed-generative-adversarial-networks","title":"Prescribed Generative Adversarial Networks","date":"2019-10-09","arxiv_id":"1910.04302","n_code_links":2,"syntology":null},{"paper":"/paper/torchbeast-a-pytorch-platform-for-distributed","slug":"torchbeast-a-pytorch-platform-for-distributed","title":"TorchBeast: A PyTorch Platform for Distributed RL","date":"2019-10-08","arxiv_id":"1910.03552","n_code_links":3,"syntology":{"ran":6,"of":9,"n_ran_checked":2,"n_instrument":4,"unverified":3,"pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","official":{"repos":["heiner/scalable_agent"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official","unlocated"]}}},{"paper":null,"slug":"randomized-shortest-paths-with-net-flows-and","title":"Randomized Shortest Paths with Net Flows and Capacity Constraints","date":"2019-10-04","arxiv_id":"1910.01849","n_code_links":0,"syntology":null},{"paper":"/paper/quantized-reinforcement-learning-quarl","slug":"quantized-reinforcement-learning-quarl","title":"QuaRL: Quantization for Fast and Environmentally Sustainable Reinforcement Learning","date":"2019-10-02","arxiv_id":"1910.01055","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["harvard-edge/quarl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"cdot-continuous-domain-adaptation-using","title":"Forward-Backward Splitting for Optimal Transport based Problems","date":"2019-09-20","arxiv_id":"1909.11448","n_code_links":0,"syntology":null},{"paper":null,"slug":"mutual-information-regularization-in-markov","title":"Mutual-Information Regularization in Markov Decision Processes and Actor-Critic Learning","date":"2019-09-11","arxiv_id":"1909.05950","n_code_links":0,"syntology":null},{"paper":null,"slug":"geometry-aware-video-object-detection-for","title":"Geometry-Aware Video Object Detection for Static Cameras","date":"2019-09-06","arxiv_id":"1909.03140","n_code_links":0,"syntology":null},{"paper":null,"slug":"conditional-vehicle-trajectories-prediction","title":"Conditional Vehicle Trajectories Prediction in CARLA Urban Environment","date":"2019-09-02","arxiv_id":"1909.00792","n_code_links":0,"syntology":null},{"paper":"/paper/vusfavariational-universal-successor-features","slug":"vusfavariational-universal-successor-features","title":"VUSFA:Variational Universal Successor Features Approximator to Improve Transfer DRL for Target Driven Visual Navigation","date":"2019-08-18","arxiv_id":"1908.06376","n_code_links":2,"syntology":null},{"paper":null,"slug":"incremental-reinforcement-learning-a-new","title":"Incremental Reinforcement Learning --- a New Continuous Reinforcement Learning Frame Based on Stochastic Differential Equation methods","date":"2019-08-08","arxiv_id":"1908.02974","n_code_links":0,"syntology":null},{"paper":"/paper/doorgym-a-scalable-door-opening-environment","slug":"doorgym-a-scalable-door-opening-environment","title":"DoorGym: A Scalable Door Opening Environment And Baseline Agent","date":"2019-08-05","arxiv_id":"1908.01887","n_code_links":1,"syntology":null},{"paper":"/paper/towards-model-based-reinforcement-learning","slug":"towards-model-based-reinforcement-learning","title":"Towards Model-based Reinforcement Learning for Industry-near Environments","date":"2019-07-27","arxiv_id":"1907.11971","n_code_links":1,"syntology":null},{"paper":"/paper/google-research-football-a-novel","slug":"google-research-football-a-novel","title":"Google Research Football: A Novel Reinforcement Learning Environment","date":"2019-07-25","arxiv_id":"1907.11180","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["google-research/football"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/unsupervised-domain-adaptation-via","slug":"unsupervised-domain-adaptation-via","title":"Unsupervised Domain Adaptation via Calibrating Uncertainties","date":"2019-07-25","arxiv_id":"1907.11202","n_code_links":1,"syntology":null},{"paper":null,"slug":"terminal-prediction-as-an-auxiliary-task-for","title":"Terminal Prediction as an Auxiliary Task for Deep Reinforcement Learning","date":"2019-07-24","arxiv_id":"1907.10827","n_code_links":0,"syntology":null},{"paper":null,"slug":"agent-modeling-as-auxiliary-task-for-deep","title":"Agent Modeling as Auxiliary Task for Deep Reinforcement Learning","date":"2019-07-22","arxiv_id":"1907.09597","n_code_links":0,"syntology":null},{"paper":"/paper/ppo-dash-improving-generalization-in-deep","slug":"ppo-dash-improving-generalization-in-deep","title":"PPO Dash: Improving Generalization in Deep Reinforcement Learning","date":"2019-07-15","arxiv_id":"1907.06704","n_code_links":1,"syntology":null},{"paper":null,"slug":"robust-guarantees-for-perception-based","title":"Robust Guarantees for Perception-Based Control","date":"2019-07-08","arxiv_id":"1907.03680","n_code_links":0,"syntology":null},{"paper":null,"slug":"modified-actor-critics","title":"Modified Actor-Critics","date":"2019-07-02","arxiv_id":"1907.01298","n_code_links":0,"syntology":null},{"paper":"/paper/end-to-end-deep-reinforcement-learning-based","slug":"end-to-end-deep-reinforcement-learning-based","title":"End-to-end Deep Reinforcement Learning Based Coreference Resolution","date":"2019-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/learning-data-augmentation-strategies-for","slug":"learning-data-augmentation-strategies-for","title":"Learning Data Augmentation Strategies for Object Detection","date":"2019-06-26","arxiv_id":"1906.11172","n_code_links":6,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tensorflow/tpu"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"neural-proximaltrust-region-policy","title":"Neural Proximal/Trust Region Policy Optimization Attains Globally Optimal Policy","date":"2019-06-25","arxiv_id":"1906.10306","n_code_links":0,"syntology":null},{"paper":"/paper/proximal-distilled-evolutionary-reinforcement","slug":"proximal-distilled-evolutionary-reinforcement","title":"Proximal Distilled Evolutionary Reinforcement Learning","date":"2019-06-24","arxiv_id":"1906.09807","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["crisbodnar/pderl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"multimodal-end-to-end-autonomous-driving","title":"Multimodal End-to-End Autonomous Driving","date":"2019-06-07","arxiv_id":"1906.03199","n_code_links":0,"syntology":null},{"paper":null,"slug":"rl-based-method-for-benchmarking-the","title":"RL-Based Method for Benchmarking the Adversarial Resilience and Robustness of Deep Reinforcement Learning Policies","date":"2019-06-03","arxiv_id":"1906.01110","n_code_links":0,"syntology":null},{"paper":null,"slug":"policy-search-by-target-distribution-learning","title":"Policy Search by Target Distribution Learning for Continuous Control","date":"2019-05-27","arxiv_id":"1905.11041","n_code_links":0,"syntology":null},{"paper":null,"slug":"combine-ppo-with-nes-to-improve-exploration","title":"Combine PPO with NES to Improve Exploration","date":"2019-05-23","arxiv_id":"1905.09492","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-q-learning-with-q-matrix-transfer","title":"Deep Q-Learning with Q-Matrix Transfer Learning for Novel Fire Evacuation Environment","date":"2019-05-23","arxiv_id":"1905.09673","n_code_links":0,"syntology":null},{"paper":"/paper/multimodal-3d-object-detection-from-simulated","slug":"multimodal-3d-object-detection-from-simulated","title":"Multimodal 3D Object Detection from Simulated Pretraining","date":"2019-05-19","arxiv_id":"1905.07754","n_code_links":2,"syntology":null},{"paper":"/paper/dimension-wise-importance-sampling-weight","slug":"dimension-wise-importance-sampling-weight","title":"Dimension-Wise Importance Sampling Weight Clipping for Sample-Efficient Reinforcement Learning","date":"2019-05-07","arxiv_id":"1905.02363","n_code_links":1,"syntology":{"ran":16,"of":19,"n_ran_checked":14,"n_instrument":2,"unverified":3,"pointer_only":18,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 1 honoured, 1 violated, 12 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","official":{"repos":["seungyulhan/disc"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"autonomous-air-traffic-controller-a-deep","title":"Autonomous Air Traffic Controller: A Deep Multi-Agent Reinforcement Learning Approach","date":"2019-05-02","arxiv_id":"1905.01303","n_code_links":0,"syntology":null},{"paper":null,"slug":"soft-q-learning-with-mutual-information","title":"Soft Q-Learning with Mutual-Information Regularization","date":"2019-05-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/supervised-policy-update","slug":"supervised-policy-update","title":"SUPERVISED POLICY UPDATE","date":"2019-05-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"towards-combining-on-off-policy-methods-for","title":"Towards Combining On-Off-Policy Methods for Real-World Applications","date":"2019-04-24","arxiv_id":"1904.10642","n_code_links":0,"syntology":null},{"paper":"/paper/190412622","slug":"190412622","title":"Talk Proposal: Towards the Realistic Evaluation of Evasion Attacks using CARLA","date":"2019-04-18","arxiv_id":"1904.12622","n_code_links":3,"syntology":null},{"paper":"/paper/rogue-gym-a-new-challenge-for-generalization","slug":"rogue-gym-a-new-challenge-for-generalization","title":"Rogue-Gym: A New Challenge for Generalization in Reinforcement Learning","date":"2019-04-17","arxiv_id":"1904.08129","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["kngwyu/rogue-gym","kngwyu/rogue-gym-agents-cog19"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/shapemask-learning-to-segment-novel-objects","slug":"shapemask-learning-to-segment-novel-objects","title":"ShapeMask: Learning to Segment Novel Objects by Refining Shape Priors","date":"2019-04-05","arxiv_id":"1904.03239","n_code_links":1,"syntology":null},{"paper":"/paper/jointly-pre-training-with-supervised","slug":"jointly-pre-training-with-supervised","title":"Jointly Pre-training with Supervised, Autoencoder, and Value Losses for Deep Reinforcement Learning","date":"2019-04-03","arxiv_id":"1904.02206","n_code_links":1,"syntology":null},{"paper":"/paper/truly-proximal-policy-optimization","slug":"truly-proximal-policy-optimization","title":"Truly Proximal Policy Optimization","date":"2019-03-19","arxiv_id":"1903.07940","n_code_links":1,"syntology":null},{"paper":"/paper/sample-efficient-model-free-reinforcement","slug":"sample-efficient-model-free-reinforcement","title":"Sample-Efficient Model-Free Reinforcement Learning with Off-Policy Critics","date":"2019-03-11","arxiv_id":"1903.04193","n_code_links":1,"syntology":null},{"paper":"/paper/end-to-end-driving-deploying-through","slug":"end-to-end-driving-deploying-through","title":"Visual-based Autonomous Driving Deployment from a Stochastic and Uncertainty-aware Perspective","date":"2019-03-03","arxiv_id":"1903.00821","n_code_links":1,"syntology":null},{"paper":"/paper/unsupervised-traffic-accident-detection-in","slug":"unsupervised-traffic-accident-detection-in","title":"Unsupervised Traffic Accident Detection in First-Person Videos","date":"2019-03-02","arxiv_id":"1903.00618","n_code_links":2,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["MoonBlvd/tad-IROS2019"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/trust-region-guided-proximal-policy","slug":"trust-region-guided-proximal-policy","title":"Trust Region-Guided Proximal Policy Optimization","date":"2019-01-29","arxiv_id":"1901.10314","n_code_links":2,"syntology":null},{"paper":"/paper/combinational-q-learning-for-dou-di-zhu","slug":"combinational-q-learning-for-dou-di-zhu","title":"Combinational Q-Learning for Dou Di Zhu","date":"2019-01-24","arxiv_id":"1901.08925","n_code_links":1,"syntology":null},{"paper":null,"slug":"distillation-strategies-for-proximal-policy","title":"Distillation Strategies for Proximal Policy Optimization","date":"2019-01-23","arxiv_id":"1901.08128","n_code_links":0,"syntology":null},{"paper":"/paper/on-policy-trust-region-policy-optimisation","slug":"on-policy-trust-region-policy-optimisation","title":"On-Policy Trust Region Policy Optimisation with Replay Buffers","date":"2019-01-18","arxiv_id":"1901.06212","n_code_links":2,"syntology":null},{"paper":null,"slug":"exploring-applications-of-deep-reinforcement","title":"Exploring applications of deep reinforcement learning for real-world autonomous driving systems","date":"2019-01-06","arxiv_id":"1901.01536","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-logarithmic-barrier-method-for-proximal","title":"A Logarithmic Barrier Method For Proximal Policy Optimization","date":"2018-12-16","arxiv_id":"1812.06502","n_code_links":0,"syntology":null},{"paper":"/paper/sada-semantic-adversarial-diagnostic-attacks","slug":"sada-semantic-adversarial-diagnostic-attacks","title":"SADA: Semantic Adversarial Diagnostic Attacks for Autonomous Applications","date":"2018-12-05","arxiv_id":"1812.02132","n_code_links":1,"syntology":null},{"paper":null,"slug":"exploration-versus-exploitation-in","title":"Exploration versus exploitation in reinforcement learning: a stochastic control approach","date":"2018-12-04","arxiv_id":"1812.01552","n_code_links":0,"syntology":null},{"paper":null,"slug":"using-monte-carlo-tree-search-as-a","title":"Using Monte Carlo Tree Search as a Demonstrator within Asynchronous Deep RL","date":"2018-11-30","arxiv_id":"1812.00045","n_code_links":0,"syntology":null},{"paper":"/paper/single-agent-policy-tree-search-with","slug":"single-agent-policy-tree-search-with","title":"Single-Agent Policy Tree Search With Guarantees","date":"2018-11-27","arxiv_id":"1811.10928","n_code_links":1,"syntology":null},{"paper":"/paper/universal-semi-supervised-semantic","slug":"universal-semi-supervised-semantic","title":"Universal Semi-Supervised Semantic Segmentation","date":"2018-11-26","arxiv_id":"1811.10323","n_code_links":1,"syntology":null},{"paper":null,"slug":"policy-optimization-with-model-based","title":"Policy Optimization with Model-based Explorations","date":"2018-11-18","arxiv_id":"1811.07350","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-complexity-of-exploration-in-goal","title":"On the Complexity of Exploration in Goal-Driven Navigation","date":"2018-11-16","arxiv_id":"1811.06889","n_code_links":0,"syntology":null},{"paper":"/paper/trolleymod-v10-an-open-source-simulation-and","slug":"trolleymod-v10-an-open-source-simulation-and","title":"TrolleyMod v1.0: An Open-Source Simulation and Data-Collection Platform for Ethical Decision Making in Autonomous Vehicles","date":"2018-11-14","arxiv_id":"1811.05594","n_code_links":1,"syntology":null},{"paper":null,"slug":"equivalent-constraints-for-two-view-geometry","title":"Equivalent Constraints for Two-View Geometry: Pose Solution/Pure Rotation Identification and 3D Reconstruction","date":"2018-10-13","arxiv_id":"1810.05863","n_code_links":0,"syntology":null},{"paper":"/paper/entropic-gans-meet-vaes-a-statistical","slug":"entropic-gans-meet-vaes-a-statistical","title":"Entropic GANs meet VAEs: A Statistical Approach to Compute Sample Likelihoods in GANs","date":"2018-10-09","arxiv_id":"1810.04147","n_code_links":1,"syntology":null},{"paper":"/paper/nsga-net-a-multi-objective-genetic-algorithm","slug":"nsga-net-a-multi-objective-genetic-algorithm","title":"NSGA-Net: Neural Architecture Search using Multi-Objective Genetic Algorithm","date":"2018-10-08","arxiv_id":"1810.03522","n_code_links":2,"syntology":null},{"paper":"/paper/ppo-cma-proximal-policy-optimization-with","slug":"ppo-cma-proximal-policy-optimization-with","title":"PPO-CMA: Proximal Policy Optimization with Covariance Matrix Adaptation","date":"2018-10-05","arxiv_id":"1810.02541","n_code_links":1,"syntology":null},{"paper":"/paper/reinforcement-learning-with-perturbed-rewards","slug":"reinforcement-learning-with-perturbed-rewards","title":"Reinforcement Learning with Perturbed Rewards","date":"2018-10-02","arxiv_id":"1810.01032","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":3,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["wangjksjtu/rl-perturbed-reward"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"real-time-dynamic-object-detection-for","title":"Real-time Dynamic Object Detection for Autonomous Driving using Prior 3D-Maps","date":"2018-09-28","arxiv_id":"1809.11036","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-fast-globally-linearly-convergent-algorithm","title":"A Fast Globally Linearly Convergent Algorithm for the Computation of Wasserstein Barycenters","date":"2018-09-12","arxiv_id":"1809.04249","n_code_links":0,"syntology":null},{"paper":"/paper/learning-end-to-end-autonomous-driving-using","slug":"learning-end-to-end-autonomous-driving-using","title":"Learning End-to-end Autonomous Driving using Guided Auxiliary Supervision","date":"2018-08-30","arxiv_id":"1808.10393","n_code_links":1,"syntology":null},{"paper":"/paper/adversarial-deep-reinforcement-learning-in","slug":"adversarial-deep-reinforcement-learning-in","title":"Adversarial Deep Reinforcement Learning in Portfolio Management","date":"2018-08-29","arxiv_id":"1808.09940","n_code_links":5,"syntology":null},{"paper":null,"slug":"proximal-policy-optimization-and-its-dynamic","title":"Proximal Policy Optimization and its Dynamic Version for Sequence Generation","date":"2018-08-24","arxiv_id":"1808.07982","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-optimal-algorithm-for-stochastic-and","title":"Tsallis-INF: An Optimal Algorithm for Stochastic and Adversarial Bandits","date":"2018-07-19","arxiv_id":"1807.07623","n_code_links":0,"syntology":null},{"paper":null,"slug":"gradient-band-based-adversarial-training-for","title":"Gradient Band-based Adversarial Training for Generalized Attack Immunity of A3C Path Finding","date":"2018-07-18","arxiv_id":"1807.06752","n_code_links":0,"syntology":null},{"paper":"/paper/policy-optimization-with-penalized-point","slug":"policy-optimization-with-penalized-point","title":"Policy Optimization With Penalized Point Probability Distance: An Alternative To Proximal Policy Optimization","date":"2018-07-02","arxiv_id":"1807.00442","n_code_links":2,"syntology":null},{"paper":"/paper/conditional-affordance-learning-for-driving","slug":"conditional-affordance-learning-for-driving","title":"Conditional Affordance Learning for Driving in Urban Environments","date":"2018-06-18","arxiv_id":"1806.06498","n_code_links":1,"syntology":{"ran":12,"of":16,"n_ran_checked":12,"n_instrument":0,"unverified":4,"pointer_only":1,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 1 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["xl-sr/CAL"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/semantic-road-layout-understanding-by","slug":"semantic-road-layout-understanding-by","title":"Semantic Road Layout Understanding by Generative Adversarial Inpainting","date":"2018-05-29","arxiv_id":"1805.11746","n_code_links":0,"syntology":null},{"paper":"/paper/supervised-policy-update-for-deep","slug":"supervised-policy-update-for-deep","title":"Supervised Policy Update for Deep Reinforcement Learning","date":"2018-05-29","arxiv_id":"1805.11706","n_code_links":1,"syntology":null},{"paper":"/paper/crawling-in-rogues-dungeons-with-partitioned","slug":"crawling-in-rogues-dungeons-with-partitioned","title":"Crawling in Rogue's dungeons with (partitioned) A3C","date":"2018-04-23","arxiv_id":"1804.08685","n_code_links":1,"syntology":null},{"paper":null,"slug":"an-adaptive-clipping-approach-for-proximal","title":"An Adaptive Clipping Approach for Proximal Policy Optimization","date":"2018-04-17","arxiv_id":"1804.06461","n_code_links":0,"syntology":null}],"record_sha256":"82722a1280b111ea0bd5b85d47903d4f3486a4578f3e069a938e79ddb31343bb","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}