{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/6","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":6,"pages_in_order":135,"rows_per_page":100,"rows":[501,600],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/5","next":"/task/reinforcement-learning-2/papers/7","papers":[{"url":"/paper/episodic-multi-agent-reinforcement-learning-1","slug":"episodic-multi-agent-reinforcement-learning-1","title":"Episodic Multi-agent Reinforcement Learning with Curiosity-Driven Exploration","date":"2021-11-22","arxiv_id":"2111.11032","repositories_listed":2,"syntology":null},{"url":"/paper/cleanrl-high-quality-single-file","slug":"cleanrl-high-quality-single-file","title":"CleanRL: High-quality Single-file Implementations of Deep Reinforcement Learning Algorithms","date":"2021-11-16","arxiv_id":"2111.08819","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cleanrl-high-quality-single-file#ran","syntology_url":"https://syntology.ai/paper/2111.08819","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.08819"}},"official":{"repos":["vwxyzjn/cleanrl"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/understanding-the-effects-of-dataset","slug":"understanding-the-effects-of-dataset","title":"A Dataset Perspective on Offline Reinforcement Learning","date":"2021-11-08","arxiv_id":"2111.04714","repositories_listed":2,"syntology":null},{"url":"/paper/d3rlpy-an-offline-deep-reinforcement-learning","slug":"d3rlpy-an-offline-deep-reinforcement-learning","title":"d3rlpy: An Offline Deep Reinforcement Learning Library","date":"2021-11-06","arxiv_id":"2111.03788","repositories_listed":2,"syntology":null},{"url":"/paper/context-meta-reinforcement-learning-via","slug":"context-meta-reinforcement-learning-via","title":"Context Meta-Reinforcement Learning via Neuromodulation","date":"2021-10-30","arxiv_id":"2111.00134","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/context-meta-reinforcement-learning-via#ran","syntology_url":"https://syntology.ai/paper/2111.00134","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.00134"}},"official":{"repos":["dlpbc/nm-metarl","soltoggio/ct-graph"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/intrusion-prevention-through-optimal-stopping","slug":"intrusion-prevention-through-optimal-stopping","title":"Intrusion Prevention through Optimal Stopping","date":"2021-10-30","arxiv_id":"2111.00289","repositories_listed":2,"syntology":null},{"url":"/paper/on-joint-learning-for-solving-placement-and","slug":"on-joint-learning-for-solving-placement-and","title":"On Joint Learning for Solving Placement and Routing in Chip Design","date":"2021-10-30","arxiv_id":"2111.00234","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/on-joint-learning-for-solving-placement-and#ran","syntology_url":"https://syntology.ai/paper/2111.00234","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.00234"}},"official":{"repos":["thinklab-sjtu/eda-ai"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/distributional-reinforcement-learning-for-4","slug":"distributional-reinforcement-learning-for-4","title":"Distributional Reinforcement Learning for Multi-Dimensional Reward Functions","date":"2021-10-26","arxiv_id":"2110.13578","repositories_listed":2,"syntology":null},{"url":"/paper/fault-tolerant-federated-reinforcement","slug":"fault-tolerant-federated-reinforcement","title":"Fault-Tolerant Federated Reinforcement Learning with Theoretical Guarantee","date":"2021-10-26","arxiv_id":"2110.14074","repositories_listed":2,"syntology":{"n":15,"n_ran":9,"n_constructed":3,"n_ran_checked":4,"n_instrument":5,"n_unverified":6,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":15,"phrase":"9 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 5 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/fault-tolerant-federated-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2110.14074","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.14074"}},"official":{"repos":["flint-xf-fan/Byzantine-Federeated-RL"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/model-a-modularized-end-to-end-reinforcement","slug":"model-a-modularized-end-to-end-reinforcement","title":"A Versatile and Efficient Reinforcement Learning Framework for Autonomous Driving","date":"2021-10-22","arxiv_id":"2110.11573","repositories_listed":2,"syntology":null},{"url":"/paper/cora-benchmarks-baselines-and-metrics-as-a","slug":"cora-benchmarks-baselines-and-metrics-as-a","title":"CORA: Benchmarks, Baselines, and Metrics as a Platform for Continual Reinforcement Learning Agents","date":"2021-10-19","arxiv_id":"2110.10067","repositories_listed":2,"syntology":null},{"url":"/paper/learning-temporally-consistent-1","slug":"learning-temporally-consistent-1","title":"Learning Temporally-Consistent Representations for Data-Efficient Reinforcement Learning","date":"2021-10-11","arxiv_id":"2110.04935","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":2,"n_instrument":4,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-temporally-consistent-1#ran","syntology_url":"https://syntology.ai/paper/2110.04935","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.04935"}},"official":{"repos":["anon-researcher-repo/ksl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dropout-q-functions-for-doubly-efficient","slug":"dropout-q-functions-for-doubly-efficient","title":"Dropout Q-Functions for Doubly Efficient Reinforcement Learning","date":"2021-10-05","arxiv_id":"2110.02034","repositories_listed":2,"syntology":{"n":8,"n_ran":6,"n_constructed":3,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"6 ran (of which 3 constructed an object rather than computing a result; 6 with no instrument failure: 2 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/dropout-q-functions-for-doubly-efficient#ran","syntology_url":"https://syntology.ai/paper/2110.02034","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.02034"}},"official":{"repos":["TakuyaHiraoka/Dropout-Q-Functions-for-Doubly-Efficient-Reinforcement-Learning"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/real-robot-challenge-using-deep-reinforcement","slug":"real-robot-challenge-using-deep-reinforcement","title":"Solving the Real Robot Challenge using Deep Reinforcement Learning","date":"2021-09-30","arxiv_id":"2109.15233","repositories_listed":2,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/real-robot-challenge-using-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2109.15233","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.15233"}},"official":{"repos":["robertmccarthy97/rrc_phase1"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/policy-improvement-by-planning-with-gumbel","slug":"policy-improvement-by-planning-with-gumbel","title":"Policy improvement by planning with Gumbel","date":"2021-09-29","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/metadrive-composing-diverse-driving-scenarios","slug":"metadrive-composing-diverse-driving-scenarios","title":"MetaDrive: Composing Diverse Driving Scenarios for Generalizable Reinforcement Learning","date":"2021-09-26","arxiv_id":"2109.12674","repositories_listed":2,"syntology":null},{"url":"/paper/pythia-a-customizable-hardware-prefetching","slug":"pythia-a-customizable-hardware-prefetching","title":"Pythia: A Customizable Hardware Prefetching Framework Using Online Reinforcement Learning","date":"2021-09-24","arxiv_id":"2109.12021","repositories_listed":2,"syntology":null},{"url":"/paper/marsexplorer-exploration-of-unknown-terrains","slug":"marsexplorer-exploration-of-unknown-terrains","title":"MarsExplorer: Exploration of Unknown Terrains via Deep Reinforcement Learning and Procedurally Generated Environments","date":"2021-07-21","arxiv_id":"2107.09996","repositories_listed":2,"syntology":null},{"url":"/paper/reinforcement-learning-for-optimal-stationary","slug":"reinforcement-learning-for-optimal-stationary","title":"Reinforcement Learning for Adaptive Optimal Stationary Control of Linear Stochastic Systems","date":"2021-07-16","arxiv_id":"2107.07788","repositories_listed":2,"syntology":null},{"url":"/paper/coberl-contrastive-bert-for-reinforcement","slug":"coberl-contrastive-bert-for-reinforcement","title":"CoBERL: Contrastive BERT for Reinforcement Learning","date":"2021-07-12","arxiv_id":"2107.05431","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/coberl-contrastive-bert-for-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2107.05431","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.05431"}},"official":{"repos":["deepmind/dm_control"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/crop-certifying-robust-policies-for","slug":"crop-certifying-robust-policies-for","title":"CROP: Certifying Robust Policies for Reinforcement Learning through Functional Smoothing","date":"2021-06-17","arxiv_id":"2106.09292","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":4,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"5 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/crop-certifying-robust-policies-for#ran","syntology_url":"https://syntology.ai/paper/2106.09292","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.09292"}},"official":{"repos":["ai-secure/crop"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/contrastive-reinforcement-learning-of","slug":"contrastive-reinforcement-learning-of","title":"Contrastive Reinforcement Learning of Symbolic Reasoning Domains","date":"2021-06-16","arxiv_id":"2106.09146","repositories_listed":2,"syntology":{"n":15,"n_ran":4,"n_constructed":1,"n_ran_checked":3,"n_instrument":1,"n_unverified":11,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":15,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 11 unverified","sample_list":"/paper/contrastive-reinforcement-learning-of#ran","syntology_url":"https://syntology.ai/paper/2106.09146","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.09146"}},"official":{"repos":["gpoesia/socratic-tutor"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":11,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/optical-tactile-sim-to-real-policy-transfer","slug":"optical-tactile-sim-to-real-policy-transfer","title":"Tactile Sim-to-Real Policy Transfer via Real-to-Sim Image Translation","date":"2021-06-16","arxiv_id":"2106.08796","repositories_listed":2,"syntology":null},{"url":"/paper/end-to-end-learning-of-keypoint","slug":"end-to-end-learning-of-keypoint","title":"Learning of feature points without additional supervision improves reinforcement learning from images","date":"2021-06-15","arxiv_id":"2106.07995","repositories_listed":2,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/end-to-end-learning-of-keypoint#ran","syntology_url":"https://syntology.ai/paper/2106.07995","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.07995"}},"official":{"repos":["rinuboney/fpac","rinuboney/keyq"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/recomposing-the-reinforcement-learning","slug":"recomposing-the-reinforcement-learning","title":"Recomposing the Reinforcement Learning Building Blocks with Hypernetworks","date":"2021-06-12","arxiv_id":"2106.06842","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/recomposing-the-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2106.06842","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.06842"}},"official":{"repos":["keynans/HypeRL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pebble-feedback-efficient-interactive","slug":"pebble-feedback-efficient-interactive","title":"PEBBLE: Feedback-Efficient Interactive Reinforcement Learning via Relabeling Experience and Unsupervised Pre-training","date":"2021-06-09","arxiv_id":"2106.05091","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_constructed":1,"n_ran_checked":4,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"7 ran (of which 1 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pebble-feedback-efficient-interactive#ran","syntology_url":"https://syntology.ai/paper/2106.05091","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.05091"}},"official":{"repos":["rll-research/bpref"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/playvirtual-augmenting-cycle-consistent","slug":"playvirtual-augmenting-cycle-consistent","title":"PlayVirtual: Augmenting Cycle-Consistent Virtual Trajectories for Reinforcement Learning","date":"2021-06-08","arxiv_id":"2106.04152","repositories_listed":2,"syntology":{"n":9,"n_ran":7,"n_constructed":4,"n_ran_checked":4,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"7 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/playvirtual-augmenting-cycle-consistent#ran","syntology_url":"https://syntology.ai/paper/2106.04152","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.04152"}},"official":{"repos":["microsoft/Playvirtual"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/causal-influence-detection-for-improving","slug":"causal-influence-detection-for-improving","title":"Causal Influence Detection for Improving Efficiency in Reinforcement Learning","date":"2021-06-07","arxiv_id":"2106.03443","repositories_listed":2,"syntology":{"n":4,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 4 unverified","sample_list":"/paper/causal-influence-detection-for-improving#ran","syntology_url":"https://syntology.ai/paper/2106.03443","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.03443"}},"official":{"repos":["martius-lab/cid-in-rl"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/closed-form-analytical-results-for-maximum","slug":"closed-form-analytical-results-for-maximum","title":"Entropy Regularized Reinforcement Learning Using Large Deviation Theory","date":"2021-06-07","arxiv_id":"2106.03931","repositories_listed":2,"syntology":null},{"url":"/paper/celebrating-diversity-in-shared-multi-agent","slug":"celebrating-diversity-in-shared-multi-agent","title":"Celebrating Diversity in Shared Multi-Agent Reinforcement Learning","date":"2021-06-04","arxiv_id":"2106.02195","repositories_listed":2,"syntology":null},{"url":"/paper/mico-learning-improved-representations-via","slug":"mico-learning-improved-representations-via","title":"MICo: Improved representations via sampling-based state similarity for Markov decision processes","date":"2021-06-03","arxiv_id":"2106.08229","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":1,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 1 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mico-learning-improved-representations-via#ran","syntology_url":"https://syntology.ai/paper/2106.08229","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.08229"}},"official":{"repos":["google-research/google-research"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/reinforcement-learning-as-one-big-sequence","slug":"reinforcement-learning-as-one-big-sequence","title":"Offline Reinforcement Learning as One Big Sequence Modeling Problem","date":"2021-06-03","arxiv_id":"2106.02039","repositories_listed":2,"syntology":{"n":28,"n_ran":23,"n_constructed":0,"n_ran_checked":22,"n_instrument":1,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":21,"n_pointer_only":1,"phrase":"23 ran (of which 0 constructed an object rather than computing a result; 22 with no instrument failure: 1 honoured, 0 violated, 21 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/reinforcement-learning-as-one-big-sequence#ran","syntology_url":"https://syntology.ai/paper/2106.02039","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.02039"}},"official":{"repos":["JannerM/trajectory-transformer"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":5,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/uncertainty-weighted-actor-critic-for-offline","slug":"uncertainty-weighted-actor-critic-for-offline","title":"Uncertainty Weighted Actor-Critic for Offline Reinforcement Learning","date":"2021-05-17","arxiv_id":"2105.08140","repositories_listed":2,"syntology":null},{"url":"/paper/rl-iot-towards-iot-interoperability-via","slug":"rl-iot-towards-iot-interoperability-via","title":"RL-IoT: Reinforcement Learning to Interact with IoT Devices","date":"2021-05-03","arxiv_id":"2105.00884","repositories_listed":2,"syntology":null},{"url":"/paper/mbrl-lib-a-modular-library-for-model-based","slug":"mbrl-lib-a-modular-library-for-model-based","title":"MBRL-Lib: A Modular Library for Model-based Reinforcement Learning","date":"2021-04-20","arxiv_id":"2104.10159","repositories_listed":2,"syntology":null},{"url":"/paper/action-advising-with-advice-imitation-in-deep","slug":"action-advising-with-advice-imitation-in-deep","title":"Action Advising with Advice Imitation in Deep Reinforcement Learning","date":"2021-04-17","arxiv_id":"2104.08441","repositories_listed":2,"syntology":null},{"url":"/paper/online-and-offline-reinforcement-learning-by","slug":"online-and-offline-reinforcement-learning-by","title":"Online and Offline Reinforcement Learning by Planning with a Learned Model","date":"2021-04-13","arxiv_id":"2104.06294","repositories_listed":2,"syntology":null},{"url":"/paper/cropgym-a-reinforcement-learning-environment","slug":"cropgym-a-reinforcement-learning-environment","title":"CropGym: a Reinforcement Learning Environment for Crop Management","date":"2021-04-09","arxiv_id":"2104.04326","repositories_listed":2,"syntology":null},{"url":"/paper/improving-robustness-of-deep-reinforcement","slug":"improving-robustness-of-deep-reinforcement","title":"Improving Robustness of Deep Reinforcement Learning Agents: Environment Attack based on the Critic Network","date":"2021-04-07","arxiv_id":"2104.03154","repositories_listed":2,"syntology":null},{"url":"/paper/reagent-point-cloud-registration-using","slug":"reagent-point-cloud-registration-using","title":"ReAgent: Point Cloud Registration using Imitation and Reinforcement Learning","date":"2021-03-28","arxiv_id":"2103.15231","repositories_listed":2,"syntology":null},{"url":"/paper/bellman-a-toolbox-for-model-based","slug":"bellman-a-toolbox-for-model-based","title":"Bellman: A Toolbox for Model-Based Reinforcement Learning in TensorFlow","date":"2021-03-26","arxiv_id":"2103.14407","repositories_listed":2,"syntology":null},{"url":"/paper/online-baum-welch-algorithm-for-hierarchical","slug":"online-baum-welch-algorithm-for-hierarchical","title":"Online Baum-Welch algorithm for Hierarchical Imitation Learning","date":"2021-03-22","arxiv_id":"2103.12197","repositories_listed":2,"syntology":null},{"url":"/paper/offline-reinforcement-learning-with-fisher","slug":"offline-reinforcement-learning-with-fisher","title":"Offline Reinforcement Learning with Fisher Divergence Critic Regularization","date":"2021-03-14","arxiv_id":"2103.08050","repositories_listed":2,"syntology":null},{"url":"/paper/multi-agent-reinforcement-learning-in-2","slug":"multi-agent-reinforcement-learning-in-2","title":"Multi-agent Reinforcement Learning in OpenSpiel: A Reproduction Report","date":"2021-02-27","arxiv_id":"2103.00187","repositories_listed":2,"syntology":null},{"url":"/paper/decoupling-value-and-policy-for","slug":"decoupling-value-and-policy-for","title":"Decoupling Value and Policy for Generalization in Reinforcement Learning","date":"2021-02-20","arxiv_id":"2102.10330","repositories_listed":2,"syntology":null},{"url":"/paper/scalable-bayesian-inverse-reinforcement-1","slug":"scalable-bayesian-inverse-reinforcement-1","title":"Scalable Bayesian Inverse Reinforcement Learning","date":"2021-02-12","arxiv_id":"2102.06483","repositories_listed":2,"syntology":{"n":4,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 4 unverified","sample_list":"/paper/scalable-bayesian-inverse-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/2102.06483","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.06483"}},"official":{"repos":["XanderJC/scalable-birl","vanderschaarlab/mlforhealthlabpub"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":[]}}},{"url":"/paper/multi-task-reinforcement-learning-with","slug":"multi-task-reinforcement-learning-with","title":"Multi-Task Reinforcement Learning with Context-based Representations","date":"2021-02-11","arxiv_id":"2102.06177","repositories_listed":2,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 3 unverified","sample_list":"/paper/multi-task-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2102.06177","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.06177"}},"official":{"repos":["facebookresearch/mtenv","facebookresearch/mtrl"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/improving-model-based-reinforcement-learning","slug":"improving-model-based-reinforcement-learning","title":"Improving Model-Based Reinforcement Learning with Internal State Representations through Self-Supervision","date":"2021-02-10","arxiv_id":"2102.05599","repositories_listed":2,"syntology":null},{"url":"/paper/neurogenetic-programming-framework-for","slug":"neurogenetic-programming-framework-for","title":"Neurogenetic Programming Framework for Explainable Reinforcement Learning","date":"2021-02-08","arxiv_id":"2102.04231","repositories_listed":2,"syntology":null},{"url":"/paper/deep-reinforcement-learning-with-dynamic","slug":"deep-reinforcement-learning-with-dynamic","title":"Tactical Optimism and Pessimism for Deep Reinforcement Learning","date":"2021-02-07","arxiv_id":"2102.03765","repositories_listed":2,"syntology":null},{"url":"/paper/hyperparameter-tricks-in-multi-agent","slug":"hyperparameter-tricks-in-multi-agent","title":"Rethinking the Implementation Matters in Cooperative Multi-Agent Reinforcement Learning","date":"2021-02-06","arxiv_id":"2102.03479","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hyperparameter-tricks-in-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2102.03479","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.03479"}},"official":{"repos":["hijkzzz/pymarl2"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/metrics-and-continuity-in-reinforcement","slug":"metrics-and-continuity-in-reinforcement","title":"Metrics and continuity in reinforcement learning","date":"2021-02-02","arxiv_id":"2102.01514","repositories_listed":2,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/metrics-and-continuity-in-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2102.01514","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.01514"}},"official":{"repos":["google-research/google-research"],"state":"official: harvested for another paper","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":[]}}},{"url":"/paper/contextualized-rewriting-for-text","slug":"contextualized-rewriting-for-text","title":"Contextualized Rewriting for Text Summarization","date":"2021-01-31","arxiv_id":"2102.00385","repositories_listed":2,"syntology":null},{"url":"/paper/counterfactual-state-explanations-for","slug":"counterfactual-state-explanations-for","title":"Counterfactual State Explanations for Reinforcement Learning Agents via Generative Deep Learning","date":"2021-01-29","arxiv_id":"2101.12446","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/counterfactual-state-explanations-for#ran","syntology_url":"https://syntology.ai/paper/2101.12446","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2101.12446"}},"official":{"repos":["mattolson93/counterfactual-state-explanations"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-reinforcement-learning-on-state-1","slug":"robust-reinforcement-learning-on-state-1","title":"Robust Reinforcement Learning on State Observations with Learned Optimal Adversary","date":"2021-01-21","arxiv_id":"2101.08452","repositories_listed":2,"syntology":{"n":8,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":8,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/robust-reinforcement-learning-on-state-1#ran","syntology_url":"https://syntology.ai/paper/2101.08452","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2101.08452"}},"official":{"repos":["huanzhang12/ATLA_robust_RL"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/solving-common-payoff-games-with-approximate","slug":"solving-common-payoff-games-with-approximate","title":"Solving Common-Payoff Games with Approximate Policy Iteration","date":"2021-01-11","arxiv_id":"2101.04237","repositories_listed":2,"syntology":{"n":9,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/solving-common-payoff-games-with-approximate#ran","syntology_url":"https://syntology.ai/paper/2101.04237","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2101.04237"}},"official":{"repos":["ssokota/capi","ssokota/tiny-hanabi"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/attention-actor-critic-algorithm-for-multi","slug":"attention-actor-critic-algorithm-for-multi","title":"Attention Actor-Critic algorithm for Multi-Agent Constrained Co-operative Reinforcement Learning","date":"2021-01-07","arxiv_id":"2101.02349","repositories_listed":2,"syntology":null},{"url":"/paper/reinforcement-learning-with-latent-flow-1","slug":"reinforcement-learning-with-latent-flow-1","title":"Reinforcement Learning with Latent Flow","date":"2021-01-06","arxiv_id":"2101.01857","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-with-latent-flow-1#ran","syntology_url":"https://syntology.ai/paper/2101.01857","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2101.01857"}},"official":{"repos":["WendyShang/flare","WendyShang/dqn_zoo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-for-control-of-valves","slug":"reinforcement-learning-for-control-of-valves","title":"Reinforcement Learning for Control of Valves","date":"2020-12-29","arxiv_id":"2012.14668","repositories_listed":2,"syntology":null},{"url":"/paper/augmenting-policy-learning-with-routines","slug":"augmenting-policy-learning-with-routines","title":"Augmenting Policy Learning with Routines Discovered from a Single Demonstration","date":"2020-12-23","arxiv_id":"2012.12469","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/augmenting-policy-learning-with-routines#ran","syntology_url":"https://syntology.ai/paper/2012.12469","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2012.12469"}},"official":{"repos":["sjtuytc/-AAAI21-RoutineAugmentedPolicyLearning-RAPL-","sjtuytc/AAAI21-RoutineAugmentedPolicyLearning"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-of-graph-matching","slug":"deep-reinforcement-learning-of-graph-matching","title":"Revocable Deep Reinforcement Learning with Affinity Regularization for Outlier-Robust Graph Matching","date":"2020-12-16","arxiv_id":"2012.08950","repositories_listed":2,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":1,"n_instrument":6,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/deep-reinforcement-learning-of-graph-matching#ran","syntology_url":"https://syntology.ai/paper/2012.08950","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2012.08950"}},"official":{"repos":["thinklab-sjtu/rgm","Thinklab-SJTU/awesome-ml4co"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/solving-the-lunar-lander-problem-under","slug":"solving-the-lunar-lander-problem-under","title":"Solving The Lunar Lander Problem under Uncertainty using Reinforcement Learning","date":"2020-11-24","arxiv_id":"2011.11850","repositories_listed":2,"syntology":null},{"url":"/paper/revisiting-rainbow-promoting-more-insightful","slug":"revisiting-rainbow-promoting-more-insightful","title":"Revisiting Rainbow: Promoting more Insightful and Inclusive Deep Reinforcement Learning Research","date":"2020-11-20","arxiv_id":"2011.14826","repositories_listed":2,"syntology":null},{"url":"/paper/plas-latent-action-space-for-offline","slug":"plas-latent-action-space-for-offline","title":"PLAS: Latent Action Space for Offline Reinforcement Learning","date":"2020-11-14","arxiv_id":"2011.07213","repositories_listed":2,"syntology":null},{"url":"/paper/softgym-benchmarking-deep-reinforcement","slug":"softgym-benchmarking-deep-reinforcement","title":"SoftGym: Benchmarking Deep Reinforcement Learning for Deformable Object Manipulation","date":"2020-11-14","arxiv_id":"2011.07215","repositories_listed":2,"syntology":null},{"url":"/paper/what-did-you-think-would-happen-explaining","slug":"what-did-you-think-would-happen-explaining","title":"What Did You Think Would Happen? Explaining Agent Behaviour Through Intended Outcomes","date":"2020-11-10","arxiv_id":"2011.05064","repositories_listed":2,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/what-did-you-think-would-happen-explaining#ran","syntology_url":"https://syntology.ai/paper/2011.05064","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.05064"}},"official":{"repos":["hmhyau/rl-intention"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/decentralized-structural-rnn-for-robot-crowd","slug":"decentralized-structural-rnn-for-robot-crowd","title":"Decentralized Structural-RNN for Robot Crowd Navigation with Deep Reinforcement Learning","date":"2020-11-09","arxiv_id":"2011.04820","repositories_listed":2,"syntology":null},{"url":"/paper/a-reinforcement-learning-approach-to-the","slug":"a-reinforcement-learning-approach-to-the","title":"A Reinforcement Learning Approach to the Orienteering Problem with Time Windows","date":"2020-11-07","arxiv_id":"2011.03647","repositories_listed":2,"syntology":null},{"url":"/paper/generalization-to-new-actions-in-1","slug":"generalization-to-new-actions-in-1","title":"Generalization to New Actions in Reinforcement Learning","date":"2020-11-03","arxiv_id":"2011.01928","repositories_listed":2,"syntology":{"n":21,"n_ran":18,"n_constructed":14,"n_ran_checked":16,"n_instrument":2,"n_unverified":3,"n_honours":2,"n_violates":0,"n_no_contract":14,"n_pointer_only":20,"phrase":"18 ran (of which 14 constructed an object rather than computing a result; 16 with no instrument failure: 2 honoured, 0 violated, 14 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/generalization-to-new-actions-in-1#ran","syntology_url":"https://syntology.ai/paper/2011.01928","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.01928"}},"official":{"repos":["clvrai/new-actions-rl","clvrai/create"],"state":"official (archive's flag): 18 ran","n_ran":18,"n_constructed":14,"n_ran_no_instrument_failure":16,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/exact-asymptotics-for-linear-quadratic","slug":"exact-asymptotics-for-linear-quadratic","title":"Exact Asymptotics for Linear Quadratic Adaptive Control","date":"2020-11-02","arxiv_id":"2011.01364","repositories_listed":2,"syntology":null},{"url":"/paper/recovery-rl-safe-reinforcement-learning-with","slug":"recovery-rl-safe-reinforcement-learning-with","title":"Recovery RL: Safe Reinforcement Learning with Learned Recovery Zones","date":"2020-10-29","arxiv_id":"2010.15920","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/recovery-rl-safe-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2010.15920","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.15920"}},"official":null}},{"url":"/paper/accelerating-reinforcement-learning-with-1","slug":"accelerating-reinforcement-learning-with-1","title":"Accelerating Reinforcement Learning with Learned Skill Priors","date":"2020-10-22","arxiv_id":"2010.11944","repositories_listed":2,"syntology":null},{"url":"/paper/improving-generalization-in-reinforcement","slug":"improving-generalization-in-reinforcement","title":"Improving Generalization in Reinforcement Learning with Mixture Regularization","date":"2020-10-21","arxiv_id":"2010.10814","repositories_listed":2,"syntology":null},{"url":"/paper/deepaveragers-offline-reinforcement-learning-1","slug":"deepaveragers-offline-reinforcement-learning-1","title":"DeepAveragers: Offline Reinforcement Learning by Solving Derived Non-Parametric MDPs","date":"2020-10-18","arxiv_id":"2010.08891","repositories_listed":2,"syntology":null},{"url":"/paper/hyperparameter-auto-tuning-in-self-supervised","slug":"hyperparameter-auto-tuning-in-self-supervised","title":"Hyperparameter Auto-tuning in Self-Supervised Robotic Learning","date":"2020-10-16","arxiv_id":"2010.08252","repositories_listed":2,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/hyperparameter-auto-tuning-in-self-supervised#ran","syntology_url":"https://syntology.ai/paper/2010.08252","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.08252"}},"official":{"repos":["birlrobotics/rlkit_autotune"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/robot-navigation-in-constrained-pedestrian","slug":"robot-navigation-in-constrained-pedestrian","title":"Robot Navigation in Constrained Pedestrian Environments using Reinforcement Learning","date":"2020-10-16","arxiv_id":"2010.08600","repositories_listed":2,"syntology":null},{"url":"/paper/uav-path-planning-using-global-and-local-map","slug":"uav-path-planning-using-global-and-local-map","title":"UAV Path Planning using Global and Local Map Information with Deep Reinforcement Learning","date":"2020-10-14","arxiv_id":"2010.06917","repositories_listed":2,"syntology":null},{"url":"/paper/epidemioptim-a-toolbox-for-the-optimization-1","slug":"epidemioptim-a-toolbox-for-the-optimization-1","title":"EpidemiOptim: A Toolbox for the Optimization of Control Policies in Epidemiological Models","date":"2020-10-09","arxiv_id":"2010.04452","repositories_listed":2,"syntology":null},{"url":"/paper/fork-a-forward-looking-actor-for-model-free-1","slug":"fork-a-forward-looking-actor-for-model-free-1","title":"FORK: A Forward-Looking Actor For Model-Free Reinforcement Learning","date":"2020-10-04","arxiv_id":"2010.01652","repositories_listed":2,"syntology":null},{"url":"/paper/pettingzoo-gym-for-multi-agent-reinforcement","slug":"pettingzoo-gym-for-multi-agent-reinforcement","title":"PettingZoo: Gym for Multi-Agent Reinforcement Learning","date":"2020-09-30","arxiv_id":"2009.14471","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pettingzoo-gym-for-multi-agent-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2009.14471","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.14471"}},"official":{"repos":["Farama-Foundation/PettingZoo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-for-process","slug":"deep-reinforcement-learning-for-process","title":"Deep Reinforcement Learning for Process Synthesis","date":"2020-09-23","arxiv_id":"2009.13265","repositories_listed":2,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-unknown","slug":"deep-reinforcement-learning-for-unknown","title":"Toward Deep Supervised Anomaly Detection: Reinforcement Learning from Partially Labeled Anomaly Data","date":"2020-09-15","arxiv_id":"2009.06847","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-reinforcement-learning-for-unknown#ran","syntology_url":"https://syntology.ai/paper/2009.06847","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.06847"}},"official":null}},{"url":"/paper/solving-challenging-dexterous-manipulation","slug":"solving-challenging-dexterous-manipulation","title":"Solving Challenging Dexterous Manipulation Tasks With Trajectory Optimisation and Reinforcement Learning","date":"2020-09-09","arxiv_id":"2009.05104","repositories_listed":2,"syntology":null},{"url":"/paper/market-making-with-reinforcement-learning-sac","slug":"market-making-with-reinforcement-learning-sac","title":"Market-making with reinforcement-learning (SAC)","date":"2020-08-27","arxiv_id":"2008.12275","repositories_listed":2,"syntology":null},{"url":"/paper/reinforcement-learning-with-quantum","slug":"reinforcement-learning-with-quantum","title":"Reinforcement Learning with Quantum Variational Circuits","date":"2020-08-15","arxiv_id":"2008.07524","repositories_listed":2,"syntology":null},{"url":"/paper/offline-meta-reinforcement-learning-with","slug":"offline-meta-reinforcement-learning-with","title":"Offline Meta-Reinforcement Learning with Advantage Weighting","date":"2020-08-13","arxiv_id":"2008.06043","repositories_listed":2,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/offline-meta-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2008.06043","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.06043"}},"official":{"repos":["eric-mitchell/macaw"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/trifinger-an-open-source-robot-for-learning","slug":"trifinger-an-open-source-robot-for-learning","title":"TriFinger: An Open-Source Robot for Learning Dexterity","date":"2020-08-08","arxiv_id":"2008.03596","repositories_listed":2,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/trifinger-an-open-source-robot-for-learning#ran","syntology_url":"https://syntology.ai/paper/2008.03596","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.03596"}},"official":null}},{"url":"/paper/explore-then-execute-adapting-without-rewards","slug":"explore-then-execute-adapting-without-rewards","title":"Decoupling Exploration and Exploitation for Meta-Reinforcement Learning without Sacrifices","date":"2020-08-06","arxiv_id":"2008.02790","repositories_listed":2,"syntology":null},{"url":"/paper/robust-deep-reinforcement-learning-through","slug":"robust-deep-reinforcement-learning-through","title":"Robust Deep Reinforcement Learning through Adversarial Loss","date":"2020-08-05","arxiv_id":"2008.01976","repositories_listed":2,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/robust-deep-reinforcement-learning-through#ran","syntology_url":"https://syntology.ai/paper/2008.01976","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.01976"}},"official":{"repos":["tuomaso/radial_rl","tuomaso/radial_rl_v2"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/evolve-to-control-evolution-based-soft-actor","slug":"evolve-to-control-evolution-based-soft-actor","title":"Maximum Mutation Reinforcement Learning for Scalable Control","date":"2020-07-24","arxiv_id":"2007.13690","repositories_listed":2,"syntology":null},{"url":"/paper/active-mr-k-space-sampling-with-reinforcement","slug":"active-mr-k-space-sampling-with-reinforcement","title":"Active MR k-space Sampling with Reinforcement Learning","date":"2020-07-20","arxiv_id":"2007.10469","repositories_listed":2,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/active-mr-k-space-sampling-with-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2007.10469","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.10469"}},"official":{"repos":["facebookresearch/active-mri-acquisition"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/discovering-reinforcement-learning-algorithms","slug":"discovering-reinforcement-learning-algorithms","title":"Discovering Reinforcement Learning Algorithms","date":"2020-07-17","arxiv_id":"2007.08794","repositories_listed":2,"syntology":null},{"url":"/paper/multi-task-reinforcement-learning-as-a-hidden","slug":"multi-task-reinforcement-learning-as-a-hidden","title":"Learning Robust State Abstractions for Hidden-Parameter Block MDPs","date":"2020-07-14","arxiv_id":"2007.07206","repositories_listed":2,"syntology":null},{"url":"/paper/an-equivalence-between-loss-functions-and-non","slug":"an-equivalence-between-loss-functions-and-non","title":"An Equivalence between Loss Functions and Non-Uniform Sampling in Experience Replay","date":"2020-07-12","arxiv_id":"2007.06049","repositories_listed":2,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 2 unverified","sample_list":"/paper/an-equivalence-between-loss-functions-and-non#ran","syntology_url":"https://syntology.ai/paper/2007.06049","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.06049"}},"official":{"repos":["sfujim/LAP-PAL"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/on-the-reliability-and-generalizability-of","slug":"on-the-reliability-and-generalizability-of","title":"On the Reliability and Generalizability of Brain-inspired Reinforcement Learning Algorithms","date":"2020-07-09","arxiv_id":"2007.04578","repositories_listed":2,"syntology":null},{"url":"/paper/one-policy-to-control-them-all-shared-modular-1","slug":"one-policy-to-control-them-all-shared-modular-1","title":"One Policy to Control Them All: Shared Modular Policies for Agent-Agnostic Control","date":"2020-07-09","arxiv_id":"2007.04976","repositories_listed":2,"syntology":null},{"url":"/paper/maximum-entropy-gain-exploration-for-long","slug":"maximum-entropy-gain-exploration-for-long","title":"Maximum Entropy Gain Exploration for Long Horizon Multi-goal Reinforcement Learning","date":"2020-07-06","arxiv_id":"2007.02832","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/maximum-entropy-gain-exploration-for-long#ran","syntology_url":"https://syntology.ai/paper/2007.02832","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.02832"}},"official":{"repos":["spitis/mrl"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/reward-machines-for-cooperative-multi-agent","slug":"reward-machines-for-cooperative-multi-agent","title":"Reward Machines for Cooperative Multi-Agent Reinforcement Learning","date":"2020-07-03","arxiv_id":"2007.01962","repositories_listed":2,"syntology":null},{"url":"/paper/mdp-homomorphic-networks-group-symmetries-in","slug":"mdp-homomorphic-networks-group-symmetries-in","title":"MDP Homomorphic Networks: Group Symmetries in Reinforcement Learning","date":"2020-06-30","arxiv_id":"2006.16908","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mdp-homomorphic-networks-group-symmetries-in#ran","syntology_url":"https://syntology.ai/paper/2006.16908","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.16908"}},"official":{"repos":["ElisevanderPol/mdp-homomorphic-networks"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/disk-learning-local-features-with-policy","slug":"disk-learning-local-features-with-policy","title":"DISK: Learning local features with policy gradient","date":"2020-06-24","arxiv_id":"2006.13566","repositories_listed":2,"syntology":{"n":9,"n_ran":8,"n_constructed":4,"n_ran_checked":7,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"8 ran (of which 4 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/disk-learning-local-features-with-policy#ran","syntology_url":"https://syntology.ai/paper/2006.13566","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.13566"}},"official":{"repos":["cvlab-epfl/disk"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}}],"record_sha256":"9bd45e152ee19f61d2f8a6ba472036d1d04713aee3e5d2e7d0a4f7e5ac83226f","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}