{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/7","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":7,"pages_in_order":152,"rows_per_page":100,"rows":[601,700],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/6","next":"/task/reinforcement-learning-1/papers/8","papers":[{"url":"/paper/large-batch-experience-replay","slug":"large-batch-experience-replay","title":"Large Batch Experience Replay","date":"2021-10-04","arxiv_id":"2110.01528","repositories_listed":2,"syntology":{"n":11,"n_ran":7,"n_constructed":4,"n_ran_checked":4,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"7 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/large-batch-experience-replay#ran","syntology_url":"https://syntology.ai/paper/2110.01528","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.01528"}},"official":{"repos":["sureli/laber"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/real-robot-challenge-using-deep-reinforcement","slug":"real-robot-challenge-using-deep-reinforcement","title":"Solving the Real Robot Challenge using Deep Reinforcement Learning","date":"2021-09-30","arxiv_id":"2109.15233","repositories_listed":2,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/real-robot-challenge-using-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2109.15233","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.15233"}},"official":{"repos":["robertmccarthy97/rrc_phase1"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/policy-improvement-by-planning-with-gumbel","slug":"policy-improvement-by-planning-with-gumbel","title":"Policy improvement by planning with Gumbel","date":"2021-09-29","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/metadrive-composing-diverse-driving-scenarios","slug":"metadrive-composing-diverse-driving-scenarios","title":"MetaDrive: Composing Diverse Driving Scenarios for Generalizable Reinforcement Learning","date":"2021-09-26","arxiv_id":"2109.12674","repositories_listed":2,"syntology":null},{"url":"/paper/pythia-a-customizable-hardware-prefetching","slug":"pythia-a-customizable-hardware-prefetching","title":"Pythia: A Customizable Hardware Prefetching Framework Using Online Reinforcement Learning","date":"2021-09-24","arxiv_id":"2109.12021","repositories_listed":2,"syntology":null},{"url":"/paper/tactile-grasp-refinement-using-deep","slug":"tactile-grasp-refinement-using-deep","title":"The Role of Tactile Sensing in Learning and Deploying Grasp Refinement Algorithms","date":"2021-09-23","arxiv_id":"2109.11234","repositories_listed":2,"syntology":null},{"url":"/paper/paint-transformer-feed-forward-neural","slug":"paint-transformer-feed-forward-neural","title":"Paint Transformer: Feed Forward Neural Painting with Stroke Prediction","date":"2021-08-09","arxiv_id":"2108.03798","repositories_listed":2,"syntology":null},{"url":"/paper/marsexplorer-exploration-of-unknown-terrains","slug":"marsexplorer-exploration-of-unknown-terrains","title":"MarsExplorer: Exploration of Unknown Terrains via Deep Reinforcement Learning and Procedurally Generated Environments","date":"2021-07-21","arxiv_id":"2107.09996","repositories_listed":2,"syntology":null},{"url":"/paper/reinforcement-learning-for-optimal-stationary","slug":"reinforcement-learning-for-optimal-stationary","title":"Reinforcement Learning for Adaptive Optimal Stationary Control of Linear Stochastic Systems","date":"2021-07-16","arxiv_id":"2107.07788","repositories_listed":2,"syntology":null},{"url":"/paper/teaching-agents-how-to-map-spatial-reasoning","slug":"teaching-agents-how-to-map-spatial-reasoning","title":"Teaching Agents how to Map: Spatial Reasoning for Multi-Object Navigation","date":"2021-07-13","arxiv_id":"2107.06011","repositories_listed":2,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/teaching-agents-how-to-map-spatial-reasoning#ran","syntology_url":"https://syntology.ai/paper/2107.06011","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.06011"}},"official":{"repos":["PierreMarza/teaching_agents_how_to_map"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/coberl-contrastive-bert-for-reinforcement","slug":"coberl-contrastive-bert-for-reinforcement","title":"CoBERL: Contrastive BERT for Reinforcement Learning","date":"2021-07-12","arxiv_id":"2107.05431","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/coberl-contrastive-bert-for-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2107.05431","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.05431"}},"official":{"repos":["deepmind/dm_control"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/out-of-distribution-dynamics-detection-rl","slug":"out-of-distribution-dynamics-detection-rl","title":"Out-of-Distribution Dynamics Detection: RL-Relevant Benchmarks and Results","date":"2021-07-11","arxiv_id":"2107.04982","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":1,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/out-of-distribution-dynamics-detection-rl#ran","syntology_url":"https://syntology.ai/paper/2107.04982","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.04982"}},"official":{"repos":["modanesh/anomalous_rl_envs","modanesh/recurrent_implicit_quantile_networks"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dora-toward-policy-optimization-for-task","slug":"dora-toward-policy-optimization-for-task","title":"DORA: Toward Policy Optimization for Task-oriented Dialogue System with Efficient Context","date":"2021-07-07","arxiv_id":"2107.03286","repositories_listed":2,"syntology":null},{"url":"/paper/crop-certifying-robust-policies-for","slug":"crop-certifying-robust-policies-for","title":"CROP: Certifying Robust Policies for Reinforcement Learning through Functional Smoothing","date":"2021-06-17","arxiv_id":"2106.09292","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":4,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"5 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/crop-certifying-robust-policies-for#ran","syntology_url":"https://syntology.ai/paper/2106.09292","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.09292"}},"official":{"repos":["ai-secure/crop"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/contrastive-reinforcement-learning-of","slug":"contrastive-reinforcement-learning-of","title":"Contrastive Reinforcement Learning of Symbolic Reasoning Domains","date":"2021-06-16","arxiv_id":"2106.09146","repositories_listed":2,"syntology":{"n":15,"n_ran":4,"n_constructed":1,"n_ran_checked":3,"n_instrument":1,"n_unverified":11,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":15,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 11 unverified","sample_list":"/paper/contrastive-reinforcement-learning-of#ran","syntology_url":"https://syntology.ai/paper/2106.09146","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.09146"}},"official":{"repos":["gpoesia/socratic-tutor"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":11,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/optical-tactile-sim-to-real-policy-transfer","slug":"optical-tactile-sim-to-real-policy-transfer","title":"Tactile Sim-to-Real Policy Transfer via Real-to-Sim Image Translation","date":"2021-06-16","arxiv_id":"2106.08796","repositories_listed":2,"syntology":null},{"url":"/paper/end-to-end-learning-of-keypoint","slug":"end-to-end-learning-of-keypoint","title":"Learning of feature points without additional supervision improves reinforcement learning from images","date":"2021-06-15","arxiv_id":"2106.07995","repositories_listed":2,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/end-to-end-learning-of-keypoint#ran","syntology_url":"https://syntology.ai/paper/2106.07995","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.07995"}},"official":{"repos":["rinuboney/fpac","rinuboney/keyq"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/recomposing-the-reinforcement-learning","slug":"recomposing-the-reinforcement-learning","title":"Recomposing the Reinforcement Learning Building Blocks with Hypernetworks","date":"2021-06-12","arxiv_id":"2106.06842","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/recomposing-the-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2106.06842","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.06842"}},"official":{"repos":["keynans/HypeRL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pebble-feedback-efficient-interactive","slug":"pebble-feedback-efficient-interactive","title":"PEBBLE: Feedback-Efficient Interactive Reinforcement Learning via Relabeling Experience and Unsupervised Pre-training","date":"2021-06-09","arxiv_id":"2106.05091","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_constructed":1,"n_ran_checked":4,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"7 ran (of which 1 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pebble-feedback-efficient-interactive#ran","syntology_url":"https://syntology.ai/paper/2106.05091","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.05091"}},"official":{"repos":["rll-research/bpref"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/playvirtual-augmenting-cycle-consistent","slug":"playvirtual-augmenting-cycle-consistent","title":"PlayVirtual: Augmenting Cycle-Consistent Virtual Trajectories for Reinforcement Learning","date":"2021-06-08","arxiv_id":"2106.04152","repositories_listed":2,"syntology":{"n":9,"n_ran":7,"n_constructed":4,"n_ran_checked":4,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"7 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/playvirtual-augmenting-cycle-consistent#ran","syntology_url":"https://syntology.ai/paper/2106.04152","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.04152"}},"official":{"repos":["microsoft/Playvirtual"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/causal-influence-detection-for-improving","slug":"causal-influence-detection-for-improving","title":"Causal Influence Detection for Improving Efficiency in Reinforcement Learning","date":"2021-06-07","arxiv_id":"2106.03443","repositories_listed":2,"syntology":{"n":4,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 4 unverified","sample_list":"/paper/causal-influence-detection-for-improving#ran","syntology_url":"https://syntology.ai/paper/2106.03443","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.03443"}},"official":{"repos":["martius-lab/cid-in-rl"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/closed-form-analytical-results-for-maximum","slug":"closed-form-analytical-results-for-maximum","title":"Entropy Regularized Reinforcement Learning Using Large Deviation Theory","date":"2021-06-07","arxiv_id":"2106.03931","repositories_listed":2,"syntology":null},{"url":"/paper/celebrating-diversity-in-shared-multi-agent","slug":"celebrating-diversity-in-shared-multi-agent","title":"Celebrating Diversity in Shared Multi-Agent Reinforcement Learning","date":"2021-06-04","arxiv_id":"2106.02195","repositories_listed":2,"syntology":null},{"url":"/paper/mico-learning-improved-representations-via","slug":"mico-learning-improved-representations-via","title":"MICo: Improved representations via sampling-based state similarity for Markov decision processes","date":"2021-06-03","arxiv_id":"2106.08229","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":1,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 1 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mico-learning-improved-representations-via#ran","syntology_url":"https://syntology.ai/paper/2106.08229","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.08229"}},"official":{"repos":["google-research/google-research"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/reinforcement-learning-as-one-big-sequence","slug":"reinforcement-learning-as-one-big-sequence","title":"Offline Reinforcement Learning as One Big Sequence Modeling Problem","date":"2021-06-03","arxiv_id":"2106.02039","repositories_listed":2,"syntology":{"n":28,"n_ran":23,"n_constructed":0,"n_ran_checked":22,"n_instrument":1,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":21,"n_pointer_only":1,"phrase":"23 ran (of which 0 constructed an object rather than computing a result; 22 with no instrument failure: 1 honoured, 0 violated, 21 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/reinforcement-learning-as-one-big-sequence#ran","syntology_url":"https://syntology.ai/paper/2106.02039","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.02039"}},"official":{"repos":["JannerM/trajectory-transformer"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":5,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/uncertainty-weighted-actor-critic-for-offline","slug":"uncertainty-weighted-actor-critic-for-offline","title":"Uncertainty Weighted Actor-Critic for Offline Reinforcement Learning","date":"2021-05-17","arxiv_id":"2105.08140","repositories_listed":2,"syntology":null},{"url":"/paper/an-open-source-multi-goal-reinforcement","slug":"an-open-source-multi-goal-reinforcement","title":"An Open-Source Multi-Goal Reinforcement Learning Environment for Robotic Manipulation with Pybullet","date":"2021-05-12","arxiv_id":"2105.05985","repositories_listed":2,"syntology":null},{"url":"/paper/rl-iot-towards-iot-interoperability-via","slug":"rl-iot-towards-iot-interoperability-via","title":"RL-IoT: Reinforcement Learning to Interact with IoT Devices","date":"2021-05-03","arxiv_id":"2105.00884","repositories_listed":2,"syntology":null},{"url":"/paper/mbrl-lib-a-modular-library-for-model-based","slug":"mbrl-lib-a-modular-library-for-model-based","title":"MBRL-Lib: A Modular Library for Model-based Reinforcement Learning","date":"2021-04-20","arxiv_id":"2104.10159","repositories_listed":2,"syntology":null},{"url":"/paper/action-advising-with-advice-imitation-in-deep","slug":"action-advising-with-advice-imitation-in-deep","title":"Action Advising with Advice Imitation in Deep Reinforcement Learning","date":"2021-04-17","arxiv_id":"2104.08441","repositories_listed":2,"syntology":null},{"url":"/paper/online-and-offline-reinforcement-learning-by","slug":"online-and-offline-reinforcement-learning-by","title":"Online and Offline Reinforcement Learning by Planning with a Learned Model","date":"2021-04-13","arxiv_id":"2104.06294","repositories_listed":2,"syntology":null},{"url":"/paper/cropgym-a-reinforcement-learning-environment","slug":"cropgym-a-reinforcement-learning-environment","title":"CropGym: a Reinforcement Learning Environment for Crop Management","date":"2021-04-09","arxiv_id":"2104.04326","repositories_listed":2,"syntology":null},{"url":"/paper/improving-robustness-of-deep-reinforcement","slug":"improving-robustness-of-deep-reinforcement","title":"Improving Robustness of Deep Reinforcement Learning Agents: Environment Attack based on the Critic Network","date":"2021-04-07","arxiv_id":"2104.03154","repositories_listed":2,"syntology":null},{"url":"/paper/the-value-of-planning-for-infinite-horizon","slug":"the-value-of-planning-for-infinite-horizon","title":"The Value of Planning for Infinite-Horizon Model Predictive Control","date":"2021-04-07","arxiv_id":"2104.02863","repositories_listed":2,"syntology":null},{"url":"/paper/reagent-point-cloud-registration-using","slug":"reagent-point-cloud-registration-using","title":"ReAgent: Point Cloud Registration using Imitation and Reinforcement Learning","date":"2021-03-28","arxiv_id":"2103.15231","repositories_listed":2,"syntology":null},{"url":"/paper/bellman-a-toolbox-for-model-based","slug":"bellman-a-toolbox-for-model-based","title":"Bellman: A Toolbox for Model-Based Reinforcement Learning in TensorFlow","date":"2021-03-26","arxiv_id":"2103.14407","repositories_listed":2,"syntology":null},{"url":"/paper/online-baum-welch-algorithm-for-hierarchical","slug":"online-baum-welch-algorithm-for-hierarchical","title":"Online Baum-Welch algorithm for Hierarchical Imitation Learning","date":"2021-03-22","arxiv_id":"2103.12197","repositories_listed":2,"syntology":null},{"url":"/paper/gym-anm-reinforcement-learning-environments","slug":"gym-anm-reinforcement-learning-environments","title":"Gym-ANM: Reinforcement Learning Environments for Active Network Management Tasks in Electricity Distribution Systems","date":"2021-03-14","arxiv_id":"2103.07932","repositories_listed":2,"syntology":null},{"url":"/paper/offline-reinforcement-learning-with-fisher","slug":"offline-reinforcement-learning-with-fisher","title":"Offline Reinforcement Learning with Fisher Divergence Critic Regularization","date":"2021-03-14","arxiv_id":"2103.08050","repositories_listed":2,"syntology":null},{"url":"/paper/continuous-coordination-as-a-realistic","slug":"continuous-coordination-as-a-realistic","title":"Continuous Coordination As a Realistic Scenario for Lifelong Learning","date":"2021-03-04","arxiv_id":"2103.03216","repositories_listed":2,"syntology":null},{"url":"/paper/multi-agent-reinforcement-learning-in-2","slug":"multi-agent-reinforcement-learning-in-2","title":"Multi-agent Reinforcement Learning in OpenSpiel: A Reproduction Report","date":"2021-02-27","arxiv_id":"2103.00187","repositories_listed":2,"syntology":null},{"url":"/paper/mixed-policy-gradient","slug":"mixed-policy-gradient","title":"Mixed Policy Gradient: off-policy reinforcement learning driven jointly by data and model","date":"2021-02-23","arxiv_id":"2102.11513","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mixed-policy-gradient#ran","syntology_url":"https://syntology.ai/paper/2102.11513","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.11513"}},"official":null}},{"url":"/paper/decoupling-value-and-policy-for","slug":"decoupling-value-and-policy-for","title":"Decoupling Value and Policy for Generalization in Reinforcement Learning","date":"2021-02-20","arxiv_id":"2102.10330","repositories_listed":2,"syntology":null},{"url":"/paper/state-entropy-maximization-with-random","slug":"state-entropy-maximization-with-random","title":"State Entropy Maximization with Random Encoders for Efficient Exploration","date":"2021-02-18","arxiv_id":"2102.09430","repositories_listed":2,"syntology":null},{"url":"/paper/scalable-bayesian-inverse-reinforcement-1","slug":"scalable-bayesian-inverse-reinforcement-1","title":"Scalable Bayesian Inverse Reinforcement Learning","date":"2021-02-12","arxiv_id":"2102.06483","repositories_listed":2,"syntology":{"n":4,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 4 unverified","sample_list":"/paper/scalable-bayesian-inverse-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/2102.06483","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.06483"}},"official":{"repos":["XanderJC/scalable-birl","vanderschaarlab/mlforhealthlabpub"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":[]}}},{"url":"/paper/multi-task-reinforcement-learning-with","slug":"multi-task-reinforcement-learning-with","title":"Multi-Task Reinforcement Learning with Context-based Representations","date":"2021-02-11","arxiv_id":"2102.06177","repositories_listed":2,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 3 unverified","sample_list":"/paper/multi-task-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2102.06177","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.06177"}},"official":{"repos":["facebookresearch/mtenv","facebookresearch/mtrl"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/improving-model-based-reinforcement-learning","slug":"improving-model-based-reinforcement-learning","title":"Improving Model-Based Reinforcement Learning with Internal State Representations through Self-Supervision","date":"2021-02-10","arxiv_id":"2102.05599","repositories_listed":2,"syntology":null},{"url":"/paper/neurogenetic-programming-framework-for","slug":"neurogenetic-programming-framework-for","title":"Neurogenetic Programming Framework for Explainable Reinforcement Learning","date":"2021-02-08","arxiv_id":"2102.04231","repositories_listed":2,"syntology":null},{"url":"/paper/deep-reinforcement-learning-with-dynamic","slug":"deep-reinforcement-learning-with-dynamic","title":"Tactical Optimism and Pessimism for Deep Reinforcement Learning","date":"2021-02-07","arxiv_id":"2102.03765","repositories_listed":2,"syntology":null},{"url":"/paper/hyperparameter-tricks-in-multi-agent","slug":"hyperparameter-tricks-in-multi-agent","title":"Rethinking the Implementation Matters in Cooperative Multi-Agent Reinforcement Learning","date":"2021-02-06","arxiv_id":"2102.03479","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hyperparameter-tricks-in-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2102.03479","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.03479"}},"official":{"repos":["hijkzzz/pymarl2"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/revisiting-prioritized-experience-replay-a-1","slug":"revisiting-prioritized-experience-replay-a-1","title":"Revisiting Prioritized Experience Replay: A Value Perspective","date":"2021-02-05","arxiv_id":"2102.03261","repositories_listed":2,"syntology":null},{"url":"/paper/metrics-and-continuity-in-reinforcement","slug":"metrics-and-continuity-in-reinforcement","title":"Metrics and continuity in reinforcement learning","date":"2021-02-02","arxiv_id":"2102.01514","repositories_listed":2,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/metrics-and-continuity-in-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2102.01514","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.01514"}},"official":{"repos":["google-research/google-research"],"state":"official: harvested for another paper","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":[]}}},{"url":"/paper/contextualized-rewriting-for-text","slug":"contextualized-rewriting-for-text","title":"Contextualized Rewriting for Text Summarization","date":"2021-01-31","arxiv_id":"2102.00385","repositories_listed":2,"syntology":null},{"url":"/paper/counterfactual-state-explanations-for","slug":"counterfactual-state-explanations-for","title":"Counterfactual State Explanations for Reinforcement Learning Agents via Generative Deep Learning","date":"2021-01-29","arxiv_id":"2101.12446","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/counterfactual-state-explanations-for#ran","syntology_url":"https://syntology.ai/paper/2101.12446","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2101.12446"}},"official":{"repos":["mattolson93/counterfactual-state-explanations"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-reinforcement-learning-on-state-1","slug":"robust-reinforcement-learning-on-state-1","title":"Robust Reinforcement Learning on State Observations with Learned Optimal Adversary","date":"2021-01-21","arxiv_id":"2101.08452","repositories_listed":2,"syntology":{"n":8,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":8,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/robust-reinforcement-learning-on-state-1#ran","syntology_url":"https://syntology.ai/paper/2101.08452","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2101.08452"}},"official":{"repos":["huanzhang12/ATLA_robust_RL"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/solving-common-payoff-games-with-approximate","slug":"solving-common-payoff-games-with-approximate","title":"Solving Common-Payoff Games with Approximate Policy Iteration","date":"2021-01-11","arxiv_id":"2101.04237","repositories_listed":2,"syntology":{"n":9,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/solving-common-payoff-games-with-approximate#ran","syntology_url":"https://syntology.ai/paper/2101.04237","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2101.04237"}},"official":{"repos":["ssokota/capi","ssokota/tiny-hanabi"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/attention-actor-critic-algorithm-for-multi","slug":"attention-actor-critic-algorithm-for-multi","title":"Attention Actor-Critic algorithm for Multi-Agent Constrained Co-operative Reinforcement Learning","date":"2021-01-07","arxiv_id":"2101.02349","repositories_listed":2,"syntology":null},{"url":"/paper/reinforcement-learning-with-latent-flow-1","slug":"reinforcement-learning-with-latent-flow-1","title":"Reinforcement Learning with Latent Flow","date":"2021-01-06","arxiv_id":"2101.01857","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-with-latent-flow-1#ran","syntology_url":"https://syntology.ai/paper/2101.01857","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2101.01857"}},"official":{"repos":["WendyShang/flare","WendyShang/dqn_zoo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-for-control-of-valves","slug":"reinforcement-learning-for-control-of-valves","title":"Reinforcement Learning for Control of Valves","date":"2020-12-29","arxiv_id":"2012.14668","repositories_listed":2,"syntology":null},{"url":"/paper/augmenting-policy-learning-with-routines","slug":"augmenting-policy-learning-with-routines","title":"Augmenting Policy Learning with Routines Discovered from a Single Demonstration","date":"2020-12-23","arxiv_id":"2012.12469","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/augmenting-policy-learning-with-routines#ran","syntology_url":"https://syntology.ai/paper/2012.12469","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2012.12469"}},"official":{"repos":["sjtuytc/-AAAI21-RoutineAugmentedPolicyLearning-RAPL-","sjtuytc/AAAI21-RoutineAugmentedPolicyLearning"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-of-graph-matching","slug":"deep-reinforcement-learning-of-graph-matching","title":"Revocable Deep Reinforcement Learning with Affinity Regularization for Outlier-Robust Graph Matching","date":"2020-12-16","arxiv_id":"2012.08950","repositories_listed":2,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":1,"n_instrument":6,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/deep-reinforcement-learning-of-graph-matching#ran","syntology_url":"https://syntology.ai/paper/2012.08950","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2012.08950"}},"official":{"repos":["thinklab-sjtu/rgm","Thinklab-SJTU/awesome-ml4co"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/solving-the-lunar-lander-problem-under","slug":"solving-the-lunar-lander-problem-under","title":"Solving The Lunar Lander Problem under Uncertainty using Reinforcement Learning","date":"2020-11-24","arxiv_id":"2011.11850","repositories_listed":2,"syntology":null},{"url":"/paper/revisiting-rainbow-promoting-more-insightful","slug":"revisiting-rainbow-promoting-more-insightful","title":"Revisiting Rainbow: Promoting more Insightful and Inclusive Deep Reinforcement Learning Research","date":"2020-11-20","arxiv_id":"2011.14826","repositories_listed":2,"syntology":null},{"url":"/paper/plas-latent-action-space-for-offline","slug":"plas-latent-action-space-for-offline","title":"PLAS: Latent Action Space for Offline Reinforcement Learning","date":"2020-11-14","arxiv_id":"2011.07213","repositories_listed":2,"syntology":null},{"url":"/paper/softgym-benchmarking-deep-reinforcement","slug":"softgym-benchmarking-deep-reinforcement","title":"SoftGym: Benchmarking Deep Reinforcement Learning for Deformable Object Manipulation","date":"2020-11-14","arxiv_id":"2011.07215","repositories_listed":2,"syntology":null},{"url":"/paper/pymgrid-an-open-source-python-microgrid","slug":"pymgrid-an-open-source-python-microgrid","title":"pymgrid: An Open-Source Python Microgrid Simulator for Applied Artificial Intelligence Research","date":"2020-11-11","arxiv_id":"2011.08004","repositories_listed":2,"syntology":null},{"url":"/paper/what-did-you-think-would-happen-explaining","slug":"what-did-you-think-would-happen-explaining","title":"What Did You Think Would Happen? Explaining Agent Behaviour Through Intended Outcomes","date":"2020-11-10","arxiv_id":"2011.05064","repositories_listed":2,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/what-did-you-think-would-happen-explaining#ran","syntology_url":"https://syntology.ai/paper/2011.05064","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.05064"}},"official":{"repos":["hmhyau/rl-intention"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/decentralized-structural-rnn-for-robot-crowd","slug":"decentralized-structural-rnn-for-robot-crowd","title":"Decentralized Structural-RNN for Robot Crowd Navigation with Deep Reinforcement Learning","date":"2020-11-09","arxiv_id":"2011.04820","repositories_listed":2,"syntology":null},{"url":"/paper/a-reinforcement-learning-approach-to-the","slug":"a-reinforcement-learning-approach-to-the","title":"A Reinforcement Learning Approach to the Orienteering Problem with Time Windows","date":"2020-11-07","arxiv_id":"2011.03647","repositories_listed":2,"syntology":null},{"url":"/paper/generalization-to-new-actions-in-1","slug":"generalization-to-new-actions-in-1","title":"Generalization to New Actions in Reinforcement Learning","date":"2020-11-03","arxiv_id":"2011.01928","repositories_listed":2,"syntology":{"n":21,"n_ran":18,"n_constructed":14,"n_ran_checked":16,"n_instrument":2,"n_unverified":3,"n_honours":2,"n_violates":0,"n_no_contract":14,"n_pointer_only":20,"phrase":"18 ran (of which 14 constructed an object rather than computing a result; 16 with no instrument failure: 2 honoured, 0 violated, 14 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/generalization-to-new-actions-in-1#ran","syntology_url":"https://syntology.ai/paper/2011.01928","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.01928"}},"official":{"repos":["clvrai/new-actions-rl","clvrai/create"],"state":"official (archive's flag): 18 ran","n_ran":18,"n_constructed":14,"n_ran_no_instrument_failure":16,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/exact-asymptotics-for-linear-quadratic","slug":"exact-asymptotics-for-linear-quadratic","title":"Exact Asymptotics for Linear Quadratic Adaptive Control","date":"2020-11-02","arxiv_id":"2011.01364","repositories_listed":2,"syntology":null},{"url":"/paper/recovery-rl-safe-reinforcement-learning-with","slug":"recovery-rl-safe-reinforcement-learning-with","title":"Recovery RL: Safe Reinforcement Learning with Learned Recovery Zones","date":"2020-10-29","arxiv_id":"2010.15920","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/recovery-rl-safe-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2010.15920","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.15920"}},"official":null}},{"url":"/paper/learning-guidance-rewards-with-trajectory","slug":"learning-guidance-rewards-with-trajectory","title":"Learning Guidance Rewards with Trajectory-space Smoothing","date":"2020-10-23","arxiv_id":"2010.12718","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-guidance-rewards-with-trajectory#ran","syntology_url":"https://syntology.ai/paper/2010.12718","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.12718"}},"official":{"repos":["tgangwani/GuidanceRewards"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/accelerating-reinforcement-learning-with-1","slug":"accelerating-reinforcement-learning-with-1","title":"Accelerating Reinforcement Learning with Learned Skill Priors","date":"2020-10-22","arxiv_id":"2010.11944","repositories_listed":2,"syntology":null},{"url":"/paper/improving-generalization-in-reinforcement","slug":"improving-generalization-in-reinforcement","title":"Improving Generalization in Reinforcement Learning with Mixture Regularization","date":"2020-10-21","arxiv_id":"2010.10814","repositories_listed":2,"syntology":null},{"url":"/paper/visual-navigation-in-real-world-indoor","slug":"visual-navigation-in-real-world-indoor","title":"Visual Navigation in Real-World Indoor Environments Using End-to-End Deep Reinforcement Learning","date":"2020-10-21","arxiv_id":"2010.10903","repositories_listed":2,"syntology":null},{"url":"/paper/deepaveragers-offline-reinforcement-learning-1","slug":"deepaveragers-offline-reinforcement-learning-1","title":"DeepAveragers: Offline Reinforcement Learning by Solving Derived Non-Parametric MDPs","date":"2020-10-18","arxiv_id":"2010.08891","repositories_listed":2,"syntology":null},{"url":"/paper/hyperparameter-auto-tuning-in-self-supervised","slug":"hyperparameter-auto-tuning-in-self-supervised","title":"Hyperparameter Auto-tuning in Self-Supervised Robotic Learning","date":"2020-10-16","arxiv_id":"2010.08252","repositories_listed":2,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/hyperparameter-auto-tuning-in-self-supervised#ran","syntology_url":"https://syntology.ai/paper/2010.08252","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.08252"}},"official":{"repos":["birlrobotics/rlkit_autotune"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/robot-navigation-in-constrained-pedestrian","slug":"robot-navigation-in-constrained-pedestrian","title":"Robot Navigation in Constrained Pedestrian Environments using Reinforcement Learning","date":"2020-10-16","arxiv_id":"2010.08600","repositories_listed":2,"syntology":null},{"url":"/paper/uav-path-planning-using-global-and-local-map","slug":"uav-path-planning-using-global-and-local-map","title":"UAV Path Planning using Global and Local Map Information with Deep Reinforcement Learning","date":"2020-10-14","arxiv_id":"2010.06917","repositories_listed":2,"syntology":null},{"url":"/paper/measuring-visual-generalization-in-continuous-1","slug":"measuring-visual-generalization-in-continuous-1","title":"Measuring Visual Generalization in Continuous Control from Pixels","date":"2020-10-13","arxiv_id":"2010.06740","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/measuring-visual-generalization-in-continuous-1#ran","syntology_url":"https://syntology.ai/paper/2010.06740","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.06740"}},"official":{"repos":["QData/dmc_remastered","jakegrigsby/dmc_remastered"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/epidemioptim-a-toolbox-for-the-optimization-1","slug":"epidemioptim-a-toolbox-for-the-optimization-1","title":"EpidemiOptim: A Toolbox for the Optimization of Control Policies in Epidemiological Models","date":"2020-10-09","arxiv_id":"2010.04452","repositories_listed":2,"syntology":null},{"url":"/paper/text-based-rl-agents-with-commonsense","slug":"text-based-rl-agents-with-commonsense","title":"Text-based RL Agents with Commonsense Knowledge: New Challenges, Environments and Baselines","date":"2020-10-08","arxiv_id":"2010.03790","repositories_listed":2,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/text-based-rl-agents-with-commonsense#ran","syntology_url":"https://syntology.ai/paper/2010.03790","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.03790"}},"official":{"repos":["IBM/commonsense-rl"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/fork-a-forward-looking-actor-for-model-free-1","slug":"fork-a-forward-looking-actor-for-model-free-1","title":"FORK: A Forward-Looking Actor For Model-Free Reinforcement Learning","date":"2020-10-04","arxiv_id":"2010.01652","repositories_listed":2,"syntology":null},{"url":"/paper/pettingzoo-gym-for-multi-agent-reinforcement","slug":"pettingzoo-gym-for-multi-agent-reinforcement","title":"PettingZoo: Gym for Multi-Agent Reinforcement Learning","date":"2020-09-30","arxiv_id":"2009.14471","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pettingzoo-gym-for-multi-agent-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2009.14471","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.14471"}},"official":{"repos":["Farama-Foundation/PettingZoo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-for-process","slug":"deep-reinforcement-learning-for-process","title":"Deep Reinforcement Learning for Process Synthesis","date":"2020-09-23","arxiv_id":"2009.13265","repositories_listed":2,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-unknown","slug":"deep-reinforcement-learning-for-unknown","title":"Toward Deep Supervised Anomaly Detection: Reinforcement Learning from Partially Labeled Anomaly Data","date":"2020-09-15","arxiv_id":"2009.06847","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-reinforcement-learning-for-unknown#ran","syntology_url":"https://syntology.ai/paper/2009.06847","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.06847"}},"official":null}},{"url":"/paper/solving-challenging-dexterous-manipulation","slug":"solving-challenging-dexterous-manipulation","title":"Solving Challenging Dexterous Manipulation Tasks With Trajectory Optimisation and Reinforcement Learning","date":"2020-09-09","arxiv_id":"2009.05104","repositories_listed":2,"syntology":null},{"url":"/paper/ranking-policy-decisions","slug":"ranking-policy-decisions","title":"Ranking Policy Decisions","date":"2020-08-31","arxiv_id":"2008.13607","repositories_listed":2,"syntology":null},{"url":"/paper/market-making-with-reinforcement-learning-sac","slug":"market-making-with-reinforcement-learning-sac","title":"Market-making with reinforcement-learning (SAC)","date":"2020-08-27","arxiv_id":"2008.12275","repositories_listed":2,"syntology":null},{"url":"/paper/reinforcement-learning-with-quantum","slug":"reinforcement-learning-with-quantum","title":"Reinforcement Learning with Quantum Variational Circuits","date":"2020-08-15","arxiv_id":"2008.07524","repositories_listed":2,"syntology":null},{"url":"/paper/offline-meta-reinforcement-learning-with","slug":"offline-meta-reinforcement-learning-with","title":"Offline Meta-Reinforcement Learning with Advantage Weighting","date":"2020-08-13","arxiv_id":"2008.06043","repositories_listed":2,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/offline-meta-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2008.06043","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.06043"}},"official":{"repos":["eric-mitchell/macaw"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/trifinger-an-open-source-robot-for-learning","slug":"trifinger-an-open-source-robot-for-learning","title":"TriFinger: An Open-Source Robot for Learning Dexterity","date":"2020-08-08","arxiv_id":"2008.03596","repositories_listed":2,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/trifinger-an-open-source-robot-for-learning#ran","syntology_url":"https://syntology.ai/paper/2008.03596","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.03596"}},"official":null}},{"url":"/paper/explore-then-execute-adapting-without-rewards","slug":"explore-then-execute-adapting-without-rewards","title":"Decoupling Exploration and Exploitation for Meta-Reinforcement Learning without Sacrifices","date":"2020-08-06","arxiv_id":"2008.02790","repositories_listed":2,"syntology":null},{"url":"/paper/fashion-captioning-towards-generating","slug":"fashion-captioning-towards-generating","title":"Fashion Captioning: Towards Generating Accurate Descriptions with Semantic Rewards","date":"2020-08-06","arxiv_id":"2008.02693","repositories_listed":2,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":4,"n_instrument":4,"n_unverified":0,"n_honours":1,"n_violates":2,"n_no_contract":1,"n_pointer_only":3,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 2 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fashion-captioning-towards-generating#ran","syntology_url":"https://syntology.ai/paper/2008.02693","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.02693"}},"official":{"repos":["xuewyang/Fashion_Captioning"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/robust-deep-reinforcement-learning-through","slug":"robust-deep-reinforcement-learning-through","title":"Robust Deep Reinforcement Learning through Adversarial Loss","date":"2020-08-05","arxiv_id":"2008.01976","repositories_listed":2,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/robust-deep-reinforcement-learning-through#ran","syntology_url":"https://syntology.ai/paper/2008.01976","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.01976"}},"official":{"repos":["tuomaso/radial_rl","tuomaso/radial_rl_v2"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/evolve-to-control-evolution-based-soft-actor","slug":"evolve-to-control-evolution-based-soft-actor","title":"Maximum Mutation Reinforcement Learning for Scalable Control","date":"2020-07-24","arxiv_id":"2007.13690","repositories_listed":2,"syntology":null},{"url":"/paper/active-mr-k-space-sampling-with-reinforcement","slug":"active-mr-k-space-sampling-with-reinforcement","title":"Active MR k-space Sampling with Reinforcement Learning","date":"2020-07-20","arxiv_id":"2007.10469","repositories_listed":2,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/active-mr-k-space-sampling-with-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2007.10469","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.10469"}},"official":{"repos":["facebookresearch/active-mri-acquisition"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/discovering-reinforcement-learning-algorithms","slug":"discovering-reinforcement-learning-algorithms","title":"Discovering Reinforcement Learning Algorithms","date":"2020-07-17","arxiv_id":"2007.08794","repositories_listed":2,"syntology":null},{"url":"/paper/multi-task-reinforcement-learning-as-a-hidden","slug":"multi-task-reinforcement-learning-as-a-hidden","title":"Learning Robust State Abstractions for Hidden-Parameter Block MDPs","date":"2020-07-14","arxiv_id":"2007.07206","repositories_listed":2,"syntology":null}],"record_sha256":"ad472fe9bd85b46223f5a7b3d14dfdef2f4903a6f0d1b31e2680e4735f6ed349","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}