{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/entropy-regularization/papers/12","list_of":"/method/entropy-regularization","method":"Entropy Regularization","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":12,"pages_in_order":12,"rows_per_page":100,"rows":[1101,1128],"of":1128,"counts":{"archive_papers_tagged":1128,"with_a_code_link":451,"where_syntology_ran_a_sample":156,"not_listed_spam_title":0,"listed":1128,"listed_where_code_ran":156,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":129,"every_run_a_failure_of_syntologys_instrument":27,"listed_with_a_run_with_no_instrument_failure":129,"listed_every_run_a_failure_of_syntologys_instrument":27,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/entropy-regularization","prev":"/method/entropy-regularization/papers/11","next":null,"papers":[{"paper":null,"slug":"a-brandom-ian-view-of-reinforcement-learning","title":"A Brandom-ian view of Reinforcement Learning towards strong-AI","date":"2018-03-07","arxiv_id":"1803.02912","n_code_links":0,"syntology":null},{"paper":null,"slug":"variational-inference-for-policy-gradient","title":"Variational Inference for Policy Gradient","date":"2018-02-21","arxiv_id":"1802.07833","n_code_links":0,"syntology":null},{"paper":null,"slug":"sample-efficient-deep-reinforcement-learning","title":"Sample Efficient Deep Reinforcement Learning for Dialogue Systems with Large Action Spaces","date":"2018-02-11","arxiv_id":"1802.03753","n_code_links":0,"syntology":null},{"paper":"/paper/impala-scalable-distributed-deep-rl-with","slug":"impala-scalable-distributed-deep-rl-with","title":"IMPALA: Scalable Distributed Deep-RL with Importance Weighted Actor-Learner Architectures","date":"2018-02-05","arxiv_id":"1802.01561","n_code_links":24,"syntology":{"ran":16,"of":34,"n_ran_checked":10,"n_instrument":6,"unverified":18,"pointer_only":3,"phrase":"16 ran (of which 6 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 1 violated, 8 with no contract checked; 6 where Syntology's instrument failed) · 18 unverified","official":{"repos":["deepmind/scalable_agent"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"pretraining-deep-actor-critic-reinforcement","title":"Pretraining Deep Actor-Critic Reinforcement Learning Algorithms With Expert Demonstrations","date":"2018-01-31","arxiv_id":"1801.10459","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-empirical-analysis-of-proximal-policy","title":"An Empirical Analysis of Proximal Policy Optimization with Kronecker-factored Natural Gradients","date":"2018-01-17","arxiv_id":"1801.05566","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-deep-recurrent-models-with","title":"Exploring Deep Recurrent Models with Reinforcement Learning for Molecule Design","date":"2018-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/deep-neuroevolution-genetic-algorithms-are-a","slug":"deep-neuroevolution-genetic-algorithms-are-a","title":"Deep Neuroevolution: Genetic Algorithms Are a Competitive Alternative for Training Deep Neural Networks for Reinforcement Learning","date":"2017-12-18","arxiv_id":"1712.06567","n_code_links":12,"syntology":{"ran":4,"of":5,"n_ran_checked":2,"n_instrument":2,"unverified":1,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"natural-value-approximators-learning-when-to","title":"Natural Value Approximators: Learning when to Trust Past Estimates","date":"2017-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/teaching-a-machine-to-read-maps-with-deep","slug":"teaching-a-machine-to-read-maps-with-deep","title":"Teaching a Machine to Read Maps with Deep Reinforcement Learning","date":"2017-11-20","arxiv_id":"1711.07479","n_code_links":1,"syntology":null},{"paper":"/paper/carla-an-open-urban-driving-simulator","slug":"carla-an-open-urban-driving-simulator","title":"CARLA: An Open Urban Driving Simulator","date":"2017-11-10","arxiv_id":"1711.03938","n_code_links":1,"syntology":null},{"paper":null,"slug":"amber-adaptive-multi-batch-experience-replay","title":"AMBER: Adaptive Multi-Batch Experience Replay for Continuous Action Control","date":"2017-10-12","arxiv_id":"1710.04423","n_code_links":0,"syntology":null},{"paper":null,"slug":"sparse-markov-decision-processes-with-causal","title":"Sparse Markov Decision Processes with Causal Sparse Tsallis Entropy Regularization for Reinforcement Learning","date":"2017-09-19","arxiv_id":"1709.06293","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-search-through-a3c-reinforcement","title":"Improving Search through A3C Reinforcement Learning based Conversational Agent","date":"2017-09-17","arxiv_id":"1709.05638","n_code_links":0,"syntology":null},{"paper":"/paper/scalable-trust-region-method-for-deep","slug":"scalable-trust-region-method-for-deep","title":"Scalable trust-region method for deep reinforcement learning using Kronecker-factored approximation","date":"2017-08-17","arxiv_id":"1708.05144","n_code_links":8,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["openai/baselines"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/darla-improving-zero-shot-transfer-in","slug":"darla-improving-zero-shot-transfer-in","title":"DARLA: Improving Zero-Shot Transfer in Reinforcement Learning","date":"2017-07-26","arxiv_id":"1707.08475","n_code_links":1,"syntology":null},{"paper":"/paper/learning-transferable-architectures-for","slug":"learning-transferable-architectures-for","title":"Learning Transferable Architectures for Scalable Image Recognition","date":"2017-07-21","arxiv_id":"1707.07012","n_code_links":17,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":null}},{"paper":"/paper/proximal-policy-optimization-algorithms","slug":"proximal-policy-optimization-algorithms","title":"Proximal Policy Optimization Algorithms","date":"2017-07-20","arxiv_id":"1707.06347","n_code_links":188,"syntology":{"ran":99,"of":176,"n_ran_checked":71,"n_instrument":28,"unverified":77,"pointer_only":94,"phrase":"99 ran (of which 53 constructed an object rather than computing a result; 71 with no instrument failure: 7 honoured, 2 violated, 62 with no contract checked; 28 where Syntology's instrument failed) · 77 unverified","official":null}},{"paper":"/paper/noisy-networks-for-exploration","slug":"noisy-networks-for-exploration","title":"Noisy Networks for Exploration","date":"2017-06-30","arxiv_id":"1706.10295","n_code_links":15,"syntology":{"ran":1,"of":3,"n_ran_checked":0,"n_instrument":1,"unverified":2,"pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":null,"slug":"learning-to-factor-policies-and-action-value","title":"Learning to Factor Policies and Action-Value Functions: Factored Action Space Representations for Deep Reinforcement learning","date":"2017-05-20","arxiv_id":"1705.07269","n_code_links":0,"syntology":null},{"paper":"/paper/feature-control-as-intrinsic-motivation-for","slug":"feature-control-as-intrinsic-motivation-for","title":"Feature Control as Intrinsic Motivation for Hierarchical Reinforcement Learning","date":"2017-05-18","arxiv_id":"1705.06769","n_code_links":1,"syntology":null},{"paper":null,"slug":"equivalence-between-policy-gradients-and-soft","title":"Equivalence Between Policy Gradients and Soft Q-Learning","date":"2017-04-21","arxiv_id":"1704.06440","n_code_links":0,"syntology":null},{"paper":"/paper/the-reactor-a-fast-and-sample-efficient-actor","slug":"the-reactor-a-fast-and-sample-efficient-actor","title":"The Reactor: A fast and sample-efficient Actor-Critic agent for Reinforcement Learning","date":"2017-04-15","arxiv_id":"1704.04651","n_code_links":0,"syntology":null},{"paper":null,"slug":"tactics-of-adversarial-attack-on-deep","title":"Tactics of Adversarial Attack on Deep Reinforcement Learning Agents","date":"2017-03-08","arxiv_id":"1703.06748","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-policy-gradient-by-exploring-under","title":"Improving Policy Gradient by Exploring Under-appreciated Rewards","date":"2016-11-28","arxiv_id":"1611.09321","n_code_links":0,"syntology":null},{"paper":"/paper/reinforcement-learning-through-asynchronous","slug":"reinforcement-learning-through-asynchronous","title":"Reinforcement Learning through Asynchronous Advantage Actor-Critic on a GPU","date":"2016-11-18","arxiv_id":"1611.06256","n_code_links":3,"syntology":null},{"paper":"/paper/sample-efficient-actor-critic-with-experience","slug":"sample-efficient-actor-critic-with-experience","title":"Sample Efficient Actor-Critic with Experience Replay","date":"2016-11-03","arxiv_id":"1611.01224","n_code_links":7,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":null}},{"paper":"/paper/asynchronous-methods-for-deep-reinforcement","slug":"asynchronous-methods-for-deep-reinforcement","title":"Asynchronous Methods for Deep Reinforcement Learning","date":"2016-02-04","arxiv_id":"1602.01783","n_code_links":70,"syntology":{"ran":60,"of":95,"n_ran_checked":51,"n_instrument":9,"unverified":35,"pointer_only":20,"phrase":"60 ran (of which 20 constructed an object rather than computing a result; 51 with no instrument failure: 2 honoured, 1 violated, 48 with no contract checked; 9 where Syntology's instrument failed) · 35 unverified","official":null}}],"record_sha256":"81160c4a0d017ee9e4750b30963dd04cb1e3183cd0e3c32d64c0778c54821cb8","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}