{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/34","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":34,"pages_in_order":135,"rows_per_page":100,"rows":[3301,3400],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/33","next":"/task/reinforcement-learning-2/papers/35","papers":[{"url":"/paper/efficient-ridesharing-dispatch-using-multi","slug":"efficient-ridesharing-dispatch-using-multi","title":"Efficient Ridesharing Dispatch Using Multi-Agent Reinforcement Learning","date":"2020-06-18","arxiv_id":"2006.10897","repositories_listed":1,"syntology":null},{"url":"/paper/delta-schema-network-in-model-based","slug":"delta-schema-network-in-model-based","title":"Delta Schema Network in Model-based Reinforcement Learning","date":"2020-06-17","arxiv_id":"2006.09950","repositories_listed":1,"syntology":null},{"url":"/paper/forgetful-experience-replay-in-hierarchical","slug":"forgetful-experience-replay-in-hierarchical","title":"Forgetful Experience Replay in Hierarchical Reinforcement Learning from Demonstrations","date":"2020-06-17","arxiv_id":"2006.09939","repositories_listed":1,"syntology":null},{"url":"/paper/green-simulation-assisted-reinforcement","slug":"green-simulation-assisted-reinforcement","title":"Green Simulation Assisted Reinforcement Learning with Model Risk for Biomanufacturing Learning and Control","date":"2020-06-17","arxiv_id":"2006.09919","repositories_listed":1,"syntology":null},{"url":"/paper/model-based-adversarial-meta-reinforcement","slug":"model-based-adversarial-meta-reinforcement","title":"Model-based Adversarial Meta-Reinforcement Learning","date":"2020-06-16","arxiv_id":"2006.08875","repositories_listed":1,"syntology":null},{"url":"/paper/opponent-modelling-with-local-information","slug":"opponent-modelling-with-local-information","title":"Agent Modelling under Partial Observability for Deep Reinforcement Learning","date":"2020-06-16","arxiv_id":"2006.09447","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/opponent-modelling-with-local-information#ran","syntology_url":"https://syntology.ai/paper/2006.09447","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.09447"}},"official":{"repos":["uoe-agents/LIAM"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-model-based-reinforcement-learning-1","slug":"efficient-model-based-reinforcement-learning-1","title":"Efficient Model-Based Reinforcement Learning through Optimistic Policy Search and Planning","date":"2020-06-15","arxiv_id":"2006.08684","repositories_listed":1,"syntology":null},{"url":"/paper/learn-to-effectively-explore-in-context-based","slug":"learn-to-effectively-explore-in-context-based","title":"MetaCURE: Meta Reinforcement Learning with Empowerment-Driven Exploration","date":"2020-06-15","arxiv_id":"2006.08170","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learn-to-effectively-explore-in-context-based#ran","syntology_url":"https://syntology.ai/paper/2006.08170","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.08170"}},"official":{"repos":["NagisaZj/MetaCURE-Public"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multiagent-reinforcement-learning-based","slug":"multiagent-reinforcement-learning-based","title":"Multiagent Reinforcement Learning based Energy Beamforming Control","date":"2020-06-15","arxiv_id":"2006.08829","repositories_listed":1,"syntology":null},{"url":"/paper/optimistic-distributionally-robust-policy","slug":"optimistic-distributionally-robust-policy","title":"Optimistic Distributionally Robust Policy Optimization","date":"2020-06-14","arxiv_id":"2006.07815","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/optimistic-distributionally-robust-policy#ran","syntology_url":"https://syntology.ai/paper/2006.07815","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.07815"}},"official":{"repos":["kadysongbb/dr-trpo"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/exchangeable-models-in-meta-reinforcement","slug":"exchangeable-models-in-meta-reinforcement","title":"Exchangeable Models in Meta Reinforcement Learning","date":"2020-06-12","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/recurrent-sum-product-max-networks-for","slug":"recurrent-sum-product-max-networks-for","title":"Recurrent Sum-Product-Max Networks for Decision Making in Perfectly-Observed Environments","date":"2020-06-12","arxiv_id":"2006.07300","repositories_listed":1,"syntology":null},{"url":"/paper/distributed-reinforcement-learning-in-multi","slug":"distributed-reinforcement-learning-in-multi","title":"Multi-Agent Reinforcement Learning in Stochastic Networked Systems","date":"2020-06-11","arxiv_id":"2006.06555","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/distributed-reinforcement-learning-in-multi#ran","syntology_url":"https://syntology.ai/paper/2006.06555","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.06555"}},"official":null}},{"url":"/paper/reinforcement-learning-from-a-mixture-of","slug":"reinforcement-learning-from-a-mixture-of","title":"Continuous Action Reinforcement Learning from a Mixture of Interpretable Experts","date":"2020-06-10","arxiv_id":"2006.05911","repositories_listed":1,"syntology":null},{"url":"/paper/robust-detection-of-adaptive-spammers-by-nash","slug":"robust-detection-of-adaptive-spammers-by-nash","title":"Robust Spammer Detection by Nash Reinforcement Learning","date":"2020-06-10","arxiv_id":"2006.06069","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/robust-detection-of-adaptive-spammers-by-nash#ran","syntology_url":"https://syntology.ai/paper/2006.06069","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.06069"}},"official":{"repos":["YingtongDou/Nash-Detect"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/constrained-episodic-reinforcement-learning","slug":"constrained-episodic-reinforcement-learning","title":"Constrained episodic reinforcement learning in concave-convex and knapsack settings","date":"2020-06-09","arxiv_id":"2006.05051","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/constrained-episodic-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2006.05051","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.05051"}},"official":{"repos":["miryoosefi/ConRL"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/online-learning-in-iterated-prisoner-s","slug":"online-learning-in-iterated-prisoner-s","title":"Online Learning in Iterated Prisoner's Dilemma to Mimic Human Behavior","date":"2020-06-09","arxiv_id":"2006.06580","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-under-moral","slug":"reinforcement-learning-under-moral","title":"Reinforcement Learning Under Moral Uncertainty","date":"2020-06-08","arxiv_id":"2006.04734","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-under-moral#ran","syntology_url":"https://syntology.ai/paper/2006.04734","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.04734"}},"official":{"repos":["uber-research/normative-uncertainty"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dual-policy-distillation","slug":"dual-policy-distillation","title":"Dual Policy Distillation","date":"2020-06-07","arxiv_id":"2006.04061","repositories_listed":1,"syntology":{"n":17,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":17,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/dual-policy-distillation#ran","syntology_url":"https://syntology.ai/paper/2006.04061","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.04061"}},"official":null}},{"url":"/paper/reinforcement-learning-for-multi-product","slug":"reinforcement-learning-for-multi-product","title":"Reinforcement Learning for Multi-Product Multi-Node Inventory Management in Supply Chains","date":"2020-06-07","arxiv_id":"2006.04037","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/reinforcement-learning-for-multi-product#ran","syntology_url":"https://syntology.ai/paper/2006.04037","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.04037"}},"official":null}},{"url":"/paper/proximal-gradient-temporal-difference","slug":"proximal-gradient-temporal-difference","title":"Proximal Gradient Temporal Difference Learning: Stable Reinforcement Learning with Polynomial Sample Complexity","date":"2020-06-06","arxiv_id":"2006.03976","repositories_listed":1,"syntology":null},{"url":"/paper/curiosity-killed-the-cat-and-the","slug":"curiosity-killed-the-cat-and-the","title":"Curiosity Killed or Incapacitated the Cat and the Asymptotically Optimal Agent","date":"2020-06-05","arxiv_id":"2006.03357","repositories_listed":1,"syntology":null},{"url":"/paper/a-novel-update-mechanism-for-q-networks-based","slug":"a-novel-update-mechanism-for-q-networks-based","title":"A Novel Update Mechanism for Q-Networks Based On Extreme Learning Machines","date":"2020-06-04","arxiv_id":"2006.02986","repositories_listed":1,"syntology":null},{"url":"/paper/solving-hard-ai-planning-instances-using","slug":"solving-hard-ai-planning-instances-using","title":"Solving Hard AI Planning Instances Using Curriculum-Driven Deep Reinforcement Learning","date":"2020-06-04","arxiv_id":"2006.02689","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/solving-hard-ai-planning-instances-using#ran","syntology_url":"https://syntology.ai/paper/2006.02689","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.02689"}},"official":null}},{"url":"/paper/visual-transfer-for-reinforcement-learning","slug":"visual-transfer-for-reinforcement-learning","title":"Visual Transfer for Reinforcement Learning via Wasserstein Domain Confusion","date":"2020-06-04","arxiv_id":"2006.03465","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/visual-transfer-for-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2006.03465","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.03465"}},"official":null}},{"url":"/paper/interferobot-aligning-an-optical","slug":"interferobot-aligning-an-optical","title":"Interferobot: aligning an optical interferometer by a reinforcement learning agent","date":"2020-06-03","arxiv_id":"2006.02252","repositories_listed":1,"syntology":null},{"url":"/paper/combining-reinforcement-learning-and-1","slug":"combining-reinforcement-learning-and-1","title":"Combining Reinforcement Learning and Constraint Programming for Combinatorial Optimization","date":"2020-06-02","arxiv_id":"2006.01610","repositories_listed":1,"syntology":null},{"url":"/paper/diversity-actor-critic-sample-aware-entropy","slug":"diversity-actor-critic-sample-aware-entropy","title":"Diversity Actor-Critic: Sample-Aware Entropy Regularization for Sample-Efficient Exploration","date":"2020-06-02","arxiv_id":"2006.01419","repositories_listed":1,"syntology":null},{"url":"/paper/learning-optimal-environments-using-projected","slug":"learning-optimal-environments-using-projected","title":"Jointly Learning Environments and Control Policies with Projected Stochastic Gradient Ascent","date":"2020-06-02","arxiv_id":"2006.01738","repositories_listed":1,"syntology":null},{"url":"/paper/invariant-policy-optimization-towards","slug":"invariant-policy-optimization-towards","title":"Invariant Policy Optimization: Towards Stronger Generalization in Reinforcement Learning","date":"2020-06-01","arxiv_id":"2006.01096","repositories_listed":1,"syntology":null},{"url":"/paper/plangan-model-based-planning-with-sparse","slug":"plangan-model-based-planning-with-sparse","title":"PlanGAN: Model-based Planning With Sparse Rewards and Multiple Goals","date":"2020-06-01","arxiv_id":"2006.00900","repositories_listed":1,"syntology":null},{"url":"/paper/mm-ktd-multiple-model-kalman-temporal","slug":"mm-ktd-multiple-model-kalman-temporal","title":"MM-KTD: Multiple Model Kalman Temporal Differences for Reinforcement Learning","date":"2020-05-30","arxiv_id":"2006.00195","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning","slug":"reinforcement-learning","title":"Reinforcement Learning","date":"2020-05-29","arxiv_id":"2005.14419","repositories_listed":1,"syntology":null},{"url":"/paper/sim2real-for-peg-hole-insertion-with-eye-in","slug":"sim2real-for-peg-hole-insertion-with-eye-in","title":"Sim2Real for Peg-Hole Insertion with Eye-in-Hand Camera","date":"2020-05-29","arxiv_id":"2005.14401","repositories_listed":1,"syntology":null},{"url":"/paper/parameter-sharing-is-surprisingly-useful-for","slug":"parameter-sharing-is-surprisingly-useful-for","title":"Revisiting Parameter Sharing in Multi-Agent Deep Reinforcement Learning","date":"2020-05-27","arxiv_id":"2005.13625","repositories_listed":1,"syntology":null},{"url":"/paper/a-reinforcement-learning-approach-to-rare","slug":"a-reinforcement-learning-approach-to-rare","title":"A reinforcement learning approach to rare trajectory sampling","date":"2020-05-26","arxiv_id":"2005.12890","repositories_listed":1,"syntology":null},{"url":"/paper/alba-reinforcement-learning-for-video-object","slug":"alba-reinforcement-learning-for-video-object","title":"ALBA : Reinforcement Learning for Video Object Segmentation","date":"2020-05-26","arxiv_id":"2005.13039","repositories_listed":1,"syntology":null},{"url":"/paper/modeling-penetration-testing-with","slug":"modeling-penetration-testing-with","title":"Modeling Penetration Testing with Reinforcement Learning Using Capture-the-Flag Challenges: Trade-offs between Model-free Learning and A Priori Knowledge","date":"2020-05-26","arxiv_id":"2005.12632","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-discovery-of-interpretable-planning","slug":"automatic-discovery-of-interpretable-planning","title":"Automatic Discovery of Interpretable Planning Strategies","date":"2020-05-24","arxiv_id":"2005.11730","repositories_listed":1,"syntology":null},{"url":"/paper/decentralized-deep-reinforcement-learning-for","slug":"decentralized-deep-reinforcement-learning-for","title":"Decentralized Deep Reinforcement Learning for a Distributed and Adaptive Locomotion Controller of a Hexapod Robot","date":"2020-05-21","arxiv_id":"2005.11164","repositories_listed":1,"syntology":null},{"url":"/paper/novel-policy-seeking-with-constrained","slug":"novel-policy-seeking-with-constrained","title":"Novel Policy Seeking with Constrained Optimization","date":"2020-05-21","arxiv_id":"2005.10696","repositories_listed":1,"syntology":null},{"url":"/paper/ultrasound-video-summarization-using-deep","slug":"ultrasound-video-summarization-using-deep","title":"Ultrasound Video Summarization using Deep Reinforcement Learning","date":"2020-05-19","arxiv_id":"2005.09531","repositories_listed":1,"syntology":null},{"url":"/paper/lifelong-control-of-off-grid-microgrid-with","slug":"lifelong-control-of-off-grid-microgrid-with","title":"Lifelong Control of Off-grid Microgrid with Model Based Reinforcement Learning","date":"2020-05-16","arxiv_id":"2005.08006","repositories_listed":1,"syntology":null},{"url":"/paper/training-spiking-neural-networks-using","slug":"training-spiking-neural-networks-using","title":"Training spiking neural networks using reinforcement learning","date":"2020-05-12","arxiv_id":"2005.05941","repositories_listed":1,"syntology":null},{"url":"/paper/delay-aware-model-based-reinforcement","slug":"delay-aware-model-based-reinforcement","title":"Delay-Aware Model-Based Reinforcement Learning for Continuous Control","date":"2020-05-11","arxiv_id":"2005.05440","repositories_listed":1,"syntology":null},{"url":"/paper/delay-aware-multi-agent-reinforcement","slug":"delay-aware-multi-agent-reinforcement","title":"Delay-Aware Multi-Agent Reinforcement Learning for Cooperative and Competitive Environments","date":"2020-05-11","arxiv_id":"2005.05441","repositories_listed":1,"syntology":null},{"url":"/paper/mobile-robot-path-planning-in-dynamic","slug":"mobile-robot-path-planning-in-dynamic","title":"Mobile Robot Path Planning in Dynamic Environments through Globally Guided Reinforcement Learning","date":"2020-05-11","arxiv_id":"2005.05420","repositories_listed":1,"syntology":null},{"url":"/paper/allsteps-curriculum-driven-learning-of","slug":"allsteps-curriculum-driven-learning-of","title":"ALLSTEPS: Curriculum-driven Learning of Stepping Stone Skills","date":"2020-05-09","arxiv_id":"2005.04323","repositories_listed":1,"syntology":null},{"url":"/paper/carl-controllable-agent-with-reinforcement","slug":"carl-controllable-agent-with-reinforcement","title":"CARL: Controllable Agent with Reinforcement Learning for Quadruped Locomotion","date":"2020-05-07","arxiv_id":"2005.03288","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/carl-controllable-agent-with-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2005.03288","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.03288"}},"official":{"repos":["inventec-ai-center/carl-siggraph2020"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/curious-hierarchical-actor-critic","slug":"curious-hierarchical-actor-critic","title":"Curious Hierarchical Actor-Critic Reinforcement Learning","date":"2020-05-07","arxiv_id":"2005.03420","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/curious-hierarchical-actor-critic#ran","syntology_url":"https://syntology.ai/paper/2005.03420","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.03420"}},"official":{"repos":["knowledgetechnologyuhh/goal_conditioned_RL_baselines"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/plan2vec-unsupervised-representation-learning-1","slug":"plan2vec-unsupervised-representation-learning-1","title":"Plan2Vec: Unsupervised Representation Learning by Latent Plans","date":"2020-05-07","arxiv_id":"2005.03648","repositories_listed":1,"syntology":null},{"url":"/paper/supert-towards-new-frontiers-in-unsupervised","slug":"supert-towards-new-frontiers-in-unsupervised","title":"SUPERT: Towards New Frontiers in Unsupervised Evaluation Metrics for Multi-Document Summarization","date":"2020-05-07","arxiv_id":"2005.03724","repositories_listed":1,"syntology":null},{"url":"/paper/gifting-in-multi-agent-reinforcement-learning","slug":"gifting-in-multi-agent-reinforcement-learning","title":"Gifting in multi-agent reinforcement learning","date":"2020-05-05","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/off-policy-adversarial-inverse-reinforcement","slug":"off-policy-adversarial-inverse-reinforcement","title":"Off-Policy Adversarial Inverse Reinforcement Learning","date":"2020-05-03","arxiv_id":"2005.01138","repositories_listed":1,"syntology":null},{"url":"/paper/deep-symbolic-superoptimization-without-human","slug":"deep-symbolic-superoptimization-without-human","title":"Deep Symbolic Superoptimization Without Human Knowledge","date":"2020-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-collaborative-agents-with-rule","slug":"learning-collaborative-agents-with-rule","title":"Learning Collaborative Agents with Rule Guidance for Knowledge Graph Reasoning","date":"2020-05-01","arxiv_id":"2005.00571","repositories_listed":1,"syntology":null},{"url":"/paper/logic-and-the-2-simplicial-transformer-1","slug":"logic-and-the-2-simplicial-transformer-1","title":"Logic and the 2-Simplicial Transformer","date":"2020-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/option-discovery-using-deep-skill-chaining","slug":"option-discovery-using-deep-skill-chaining","title":"Option Discovery using Deep Skill Chaining","date":"2020-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/ract-toward-amortized-ranking-critical","slug":"ract-toward-amortized-ranking-critical","title":"RaCT: Toward Amortized Ranking-Critical Training For Collaborative Filtering","date":"2020-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/actor-critic-reinforcement-learning-for","slug":"actor-critic-reinforcement-learning-for","title":"Actor-Critic Reinforcement Learning for Control with Stability Guarantee","date":"2020-04-29","arxiv_id":"2004.14288","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/actor-critic-reinforcement-learning-for#ran","syntology_url":"https://syntology.ai/paper/2004.14288","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.14288"}},"official":{"repos":["hithmh/Actor-critic-with-stability-guarantee"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/deep-reinforcement-learning-with-graph-based","slug":"deep-reinforcement-learning-with-graph-based","title":"Graph-based State Representation for Deep Reinforcement Learning","date":"2020-04-29","arxiv_id":"2004.13965","repositories_listed":1,"syntology":null},{"url":"/paper/evolving-inborn-knowledge-for-fast-adaptation","slug":"evolving-inborn-knowledge-for-fast-adaptation","title":"Evolving Inborn Knowledge For Fast Adaptation in Dynamic POMDP Problems","date":"2020-04-27","arxiv_id":"2004.12846","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-navigate-the-synthetically","slug":"learning-to-navigate-the-synthetically","title":"Learning To Navigate The Synthetically Accessible Chemical Space Using Reinforcement Learning","date":"2020-04-26","arxiv_id":"2004.12485","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-generalization-with","slug":"reinforcement-learning-generalization-with","title":"Reinforcement Learning Generalization with Surprise Minimization","date":"2020-04-26","arxiv_id":"2004.12399","repositories_listed":1,"syntology":null},{"url":"/paper/self-paced-deep-reinforcement-learning","slug":"self-paced-deep-reinforcement-learning","title":"Self-Paced Deep Reinforcement Learning","date":"2020-04-24","arxiv_id":"2004.11812","repositories_listed":1,"syntology":null},{"url":"/paper/the-variational-bandwidth-bottleneck-1","slug":"the-variational-bandwidth-bottleneck-1","title":"The Variational Bandwidth Bottleneck: Stochastic Evaluation on an Information Budget","date":"2020-04-24","arxiv_id":"2004.11935","repositories_listed":1,"syntology":null},{"url":"/paper/correct-me-if-you-can-learning-from-error","slug":"correct-me-if-you-can-learning-from-error","title":"Correct Me If You Can: Learning from Error Corrections and Markings","date":"2020-04-23","arxiv_id":"2004.11222","repositories_listed":1,"syntology":null},{"url":"/paper/per-step-reward-a-new-perspective-for-risk","slug":"per-step-reward-a-new-perspective-for-risk","title":"Mean-Variance Policy Iteration for Risk-Averse Reinforcement Learning","date":"2020-04-22","arxiv_id":"2004.10888","repositories_listed":1,"syntology":null},{"url":"/paper/tactical-decision-making-in-autonomous","slug":"tactical-decision-making-in-autonomous","title":"Tactical Decision-Making in Autonomous Driving by Reinforcement Learning with Uncertainty Estimation","date":"2020-04-22","arxiv_id":"2004.10439","repositories_listed":1,"syntology":null},{"url":"/paper/energy-based-imitation-learning","slug":"energy-based-imitation-learning","title":"Energy-Based Imitation Learning","date":"2020-04-20","arxiv_id":"2004.09395","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/energy-based-imitation-learning#ran","syntology_url":"https://syntology.ai/paper/2004.09395","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.09395"}},"official":{"repos":["apexrl/EBIL-torch"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/analyzing-reinforcement-learning-benchmarks","slug":"analyzing-reinforcement-learning-benchmarks","title":"Analyzing Reinforcement Learning Benchmarks with Random Weight Guessing","date":"2020-04-16","arxiv_id":"2004.07707","repositories_listed":1,"syntology":null},{"url":"/paper/continual-reinforcement-learning-with-multi","slug":"continual-reinforcement-learning-with-multi","title":"Continual Reinforcement Learning with Multi-Timescale Replay","date":"2020-04-16","arxiv_id":"2004.07530","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/continual-reinforcement-learning-with-multi#ran","syntology_url":"https://syntology.ai/paper/2004.07530","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.07530"}},"official":{"repos":["ChristosKap/multi_timescale_replay"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/fast-template-matching-and-update-for-video","slug":"fast-template-matching-and-update-for-video","title":"Fast Template Matching and Update for Video Object Tracking and Segmentation","date":"2020-04-16","arxiv_id":"2004.07538","repositories_listed":1,"syntology":null},{"url":"/paper/marleme-a-multi-agent-reinforcement-learning","slug":"marleme-a-multi-agent-reinforcement-learning","title":"MARLeME: A Multi-Agent Reinforcement Learning Model Extraction Library","date":"2020-04-16","arxiv_id":"2004.07928","repositories_listed":1,"syntology":null},{"url":"/paper/optigan-generative-adversarial-networks-for","slug":"optigan-generative-adversarial-networks-for","title":"OptiGAN: Generative Adversarial Networks for Goal Optimized Sequence Generation","date":"2020-04-16","arxiv_id":"2004.07534","repositories_listed":1,"syntology":null},{"url":"/paper/prolog-technology-reinforcement-learning","slug":"prolog-technology-reinforcement-learning","title":"Prolog Technology Reinforcement Learning Prover","date":"2020-04-15","arxiv_id":"2004.06997","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-in-a-physics-inspired","slug":"reinforcement-learning-in-a-physics-inspired","title":"Reinforcement Learning in a Physics-Inspired Semi-Markov Environment","date":"2020-04-15","arxiv_id":"2004.07333","repositories_listed":1,"syntology":null},{"url":"/paper/a-text-based-deep-reinforcement-learning","slug":"a-text-based-deep-reinforcement-learning","title":"A Text-based Deep Reinforcement Learning Framework for Interactive Recommendation","date":"2020-04-14","arxiv_id":"2004.06651","repositories_listed":1,"syntology":null},{"url":"/paper/regret-bounds-for-kernel-based-reinforcement","slug":"regret-bounds-for-kernel-based-reinforcement","title":"Kernel-Based Reinforcement Learning: A Finite-Time Analysis","date":"2020-04-12","arxiv_id":"2004.05599","repositories_listed":1,"syntology":null},{"url":"/paper/self-punishment-and-reward-backfill-for-deep","slug":"self-punishment-and-reward-backfill-for-deep","title":"Self Punishment and Reward Backfill for Deep Q-Learning","date":"2020-04-10","arxiv_id":"2004.05002","repositories_listed":1,"syntology":null},{"url":"/paper/topological-quantum-compiling-with","slug":"topological-quantum-compiling-with","title":"Topological Quantum Compiling with Reinforcement Learning","date":"2020-04-09","arxiv_id":"2004.04743","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-transformers-in-rl","slug":"adaptive-transformers-in-rl","title":"Adaptive Transformers in RL","date":"2020-04-08","arxiv_id":"2004.03761","repositories_listed":1,"syntology":null},{"url":"/paper/learning-from-learners-adapting-reinforcement","slug":"learning-from-learners-adapting-reinforcement","title":"Learning from Learners: Adapting Reinforcement Learning Agents to be Competitive in a Card Game","date":"2020-04-08","arxiv_id":"2004.04000","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-task-oriented-dialog-policy","slug":"multi-agent-task-oriented-dialog-policy","title":"Multi-Agent Task-Oriented Dialog Policy Learning with Role-Aware Reward Decomposition","date":"2020-04-08","arxiv_id":"2004.03809","repositories_listed":1,"syntology":null},{"url":"/paper/solving-the-scalarization-issues-of-advantage","slug":"solving-the-scalarization-issues-of-advantage","title":"Solving the scalarization issues of Advantage-based Reinforcement Learning Algorithms","date":"2020-04-08","arxiv_id":"2004.04120","repositories_listed":1,"syntology":null},{"url":"/paper/an-application-of-deep-reinforcement-learning","slug":"an-application-of-deep-reinforcement-learning","title":"An Application of Deep Reinforcement Learning to Algorithmic Trading","date":"2020-04-07","arxiv_id":"2004.06627","repositories_listed":1,"syntology":null},{"url":"/paper/learning-2-opt-heuristics-for-the-traveling","slug":"learning-2-opt-heuristics-for-the-traveling","title":"Learning 2-opt Heuristics for the Traveling Salesman Problem via Deep Reinforcement Learning","date":"2020-04-03","arxiv_id":"2004.01608","repositories_listed":1,"syntology":null},{"url":"/paper/mri-reconstruction-with-interpretable-pixel","slug":"mri-reconstruction-with-interpretable-pixel","title":"MRI Reconstruction with Interpretable Pixel-Wise Operations Using Reinforcement Learning","date":"2020-04-03","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-reinforcement-learning-for-2","slug":"multi-agent-reinforcement-learning-for-2","title":"Multi-agent Reinforcement Learning for Networked System Control","date":"2020-04-03","arxiv_id":"2004.01339","repositories_listed":1,"syntology":{"n":16,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":14,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":16,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 14 unverified","sample_list":"/paper/multi-agent-reinforcement-learning-for-2#ran","syntology_url":"https://syntology.ai/paper/2004.01339","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.01339"}},"official":{"repos":["cts198859/deeprl_network"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":14,"ran_from_kinds":["official"]}}},{"url":"/paper/action-space-shaping-in-deep-reinforcement","slug":"action-space-shaping-in-deep-reinforcement","title":"Action Space Shaping in Deep Reinforcement Learning","date":"2020-04-02","arxiv_id":"2004.00980","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/action-space-shaping-in-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2004.00980","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.00980"}},"official":{"repos":["Miffyli/rl-action-space-shaping"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/information-state-embedding-in-partially","slug":"information-state-embedding-in-partially","title":"Information State Embedding in Partially Observable Cooperative Multi-Agent Reinforcement Learning","date":"2020-04-02","arxiv_id":"2004.01098","repositories_listed":1,"syntology":null},{"url":"/paper/augmented-q-imitation-learning-aqil","slug":"augmented-q-imitation-learning-aqil","title":"Augmented Q Imitation Learning (AQIL)","date":"2020-03-31","arxiv_id":"2004.00993","repositories_listed":1,"syntology":null},{"url":"/paper/exploration-in-action-space","slug":"exploration-in-action-space","title":"Exploration in Action Space","date":"2020-03-31","arxiv_id":"2004.00500","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-ask-medical-questions-using","slug":"learning-to-ask-medical-questions-using","title":"Learning to Ask Medical Questions using Reinforcement Learning","date":"2020-03-31","arxiv_id":"2004.00994","repositories_listed":1,"syntology":null},{"url":"/paper/optimising-lockdown-policies-for-epidemic","slug":"optimising-lockdown-policies-for-epidemic","title":"Optimising Lockdown Policies for Epidemic Control using Reinforcement Learning","date":"2020-03-31","arxiv_id":"2003.14093","repositories_listed":1,"syntology":null},{"url":"/paper/straight-to-the-point-fast-forwarding-videos","slug":"straight-to-the-point-fast-forwarding-videos","title":"Straight to the Point: Fast-forwarding Videos via Reinforcement Learning Using Textual Data","date":"2020-03-31","arxiv_id":"2003.14229","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-large-scale","slug":"deep-reinforcement-learning-for-large-scale","title":"Deep reinforcement learning for large-scale epidemic control","date":"2020-03-30","arxiv_id":"2003.13676","repositories_listed":1,"syntology":null},{"url":"/paper/multi-task-reinforcement-learning-with-soft","slug":"multi-task-reinforcement-learning-with-soft","title":"Multi-Task Reinforcement Learning with Soft Modularization","date":"2020-03-30","arxiv_id":"2003.13661","repositories_listed":1,"syntology":null},{"url":"/paper/suphx-mastering-mahjong-with-deep","slug":"suphx-mastering-mahjong-with-deep","title":"Suphx: Mastering Mahjong with Deep Reinforcement Learning","date":"2020-03-30","arxiv_id":"2003.13590","repositories_listed":1,"syntology":null},{"url":"/paper/obstacle-avoidance-and-navigation-utilizing","slug":"obstacle-avoidance-and-navigation-utilizing","title":"Obstacle Avoidance and Navigation Utilizing Reinforcement Learning with Reward Shaping","date":"2020-03-28","arxiv_id":"2003.12863","repositories_listed":1,"syntology":null}],"record_sha256":"05659dacdb7e939d581b999586762cd4b159ac03142e9a49e30b20338d3fb9fe","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}