{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/ran/7","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":7,"pages_in_order":15,"rows_per_page":100,"rows":[601,700],"of":1416,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1/papers/ran/1","prev":"/task/reinforcement-learning-1/papers/ran/6","next":"/task/reinforcement-learning-1/papers/ran/8","papers":[{"url":"/paper/scaling-laws-for-a-multi-agent-reinforcement","slug":"scaling-laws-for-a-multi-agent-reinforcement","title":"Scaling Laws for a Multi-Agent Reinforcement Learning Model","date":"2022-09-29","arxiv_id":"2210.00849","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scaling-laws-for-a-multi-agent-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2210.00849","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.00849"}},"official":{"repos":["orenneumann/alphazero-scaling-laws"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unsupervised-model-based-pre-training-for","slug":"unsupervised-model-based-pre-training-for","title":"Mastering the Unsupervised Reinforcement Learning Benchmark from Pixels","date":"2022-09-24","arxiv_id":"2209.12016","repositories_listed":1,"syntology":{"n":24,"n_ran":17,"n_constructed":14,"n_ran_checked":15,"n_instrument":2,"n_unverified":7,"n_honours":1,"n_violates":0,"n_no_contract":14,"n_pointer_only":0,"phrase":"17 ran (of which 14 constructed an object rather than computing a result; 15 with no instrument failure: 1 honoured, 0 violated, 14 with no contract checked; 2 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/unsupervised-model-based-pre-training-for#ran","syntology_url":"https://syntology.ai/paper/2209.12016","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.12016"}},"official":{"repos":["mazpie/mastering-urlb"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":14,"n_ran_no_instrument_failure":15,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/on-efficient-reinforcement-learning-for-full","slug":"on-efficient-reinforcement-learning-for-full","title":"On Efficient Reinforcement Learning for Full-length Game of StarCraft II","date":"2022-09-23","arxiv_id":"2209.11553","repositories_listed":2,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/on-efficient-reinforcement-learning-for-full#ran","syntology_url":"https://syntology.ai/paper/2209.11553","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.11553"}},"official":{"repos":["liuruoze/hiernet-sc2","liuruoze/mini-AlphaStar"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/honor-of-kings-arena-an-environment-for","slug":"honor-of-kings-arena-an-environment-for","title":"Honor of Kings Arena: an Environment for Generalization in Competitive Reinforcement Learning","date":"2022-09-18","arxiv_id":"2209.08483","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/honor-of-kings-arena-an-environment-for#ran","syntology_url":"https://syntology.ai/paper/2209.08483","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.08483"}},"official":{"repos":["tencent-ailab/hok_env"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/continuous-mdp-homomorphisms-and-homomorphic","slug":"continuous-mdp-homomorphisms-and-homomorphic","title":"Continuous MDP Homomorphisms and Homomorphic Policy Gradient","date":"2022-09-15","arxiv_id":"2209.07364","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":12,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/continuous-mdp-homomorphisms-and-homomorphic#ran","syntology_url":"https://syntology.ai/paper/2209.07364","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.07364"}},"official":{"repos":["sahandrez/homomorphic_policy_gradient"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-for-1","slug":"deep-reinforcement-learning-for-1","title":"Deep Reinforcement Learning for Cryptocurrency Trading: Practical Approach to Address Backtest Overfitting","date":"2022-09-12","arxiv_id":"2209.05559","repositories_listed":1,"syntology":{"n":13,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/deep-reinforcement-learning-for-1#ran","syntology_url":"https://syntology.ai/paper/2209.05559","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.05559"}},"official":null}},{"url":"/paper/white-box-adversarial-policies-in-deep","slug":"white-box-adversarial-policies-in-deep","title":"Red Teaming with Mind Reading: White-Box Adversarial Policies Against RL Agents","date":"2022-09-05","arxiv_id":"2209.02167","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/white-box-adversarial-policies-in-deep#ran","syntology_url":"https://syntology.ai/paper/2209.02167","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.02167"}},"official":{"repos":["thestephencasper/lm_white_box_attacks","thestephencasper/white_box_rarl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/transformers-are-sample-efficient-world","slug":"transformers-are-sample-efficient-world","title":"Transformers are Sample-Efficient World Models","date":"2022-09-01","arxiv_id":"2209.00588","repositories_listed":2,"syntology":{"n":26,"n_ran":17,"n_constructed":13,"n_ran_checked":16,"n_instrument":1,"n_unverified":9,"n_honours":1,"n_violates":0,"n_no_contract":15,"n_pointer_only":26,"phrase":"17 ran (of which 13 constructed an object rather than computing a result; 16 with no instrument failure: 1 honoured, 0 violated, 15 with no contract checked; 1 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/transformers-are-sample-efficient-world#ran","syntology_url":"https://syntology.ai/paper/2209.00588","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.00588"}},"official":{"repos":["eloialonso/iris"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":13,"n_ran_no_instrument_failure":16,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/style-agnostic-reinforcement-learning","slug":"style-agnostic-reinforcement-learning","title":"Style-Agnostic Reinforcement Learning","date":"2022-08-31","arxiv_id":"2208.14863","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/style-agnostic-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2208.14863","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.14863"}},"official":{"repos":["postech-cvlab/style-agnostic-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unsupervised-representation-learning-in-deep","slug":"unsupervised-representation-learning-in-deep","title":"Unsupervised Representation Learning in Deep Reinforcement Learning: A Review","date":"2022-08-27","arxiv_id":"2208.14226","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/unsupervised-representation-learning-in-deep#ran","syntology_url":"https://syntology.ai/paper/2208.14226","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.14226"}},"official":{"repos":["nicob15/state_representation_learning_methods"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/pd-morl-preference-driven-multi-objective","slug":"pd-morl-preference-driven-multi-objective","title":"PD-MORL: Preference-Driven Multi-Objective Reinforcement Learning Algorithm","date":"2022-08-16","arxiv_id":"2208.07914","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/pd-morl-preference-driven-multi-objective#ran","syntology_url":"https://syntology.ai/paper/2208.07914","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.07914"}},"official":{"repos":["tbasaklar/PDMORL-Preference-Driven-Multi-Objective-Reinforcement-Learning-Algorithm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/diffusion-policies-as-an-expressive-policy","slug":"diffusion-policies-as-an-expressive-policy","title":"Diffusion Policies as an Expressive Policy Class for Offline Reinforcement Learning","date":"2022-08-12","arxiv_id":"2208.06193","repositories_listed":3,"syntology":{"n":18,"n_ran":11,"n_constructed":5,"n_ran_checked":11,"n_instrument":0,"n_unverified":7,"n_honours":1,"n_violates":0,"n_no_contract":10,"n_pointer_only":10,"phrase":"11 ran (of which 5 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/diffusion-policies-as-an-expressive-policy#ran","syntology_url":"https://syntology.ai/paper/2208.06193","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.06193"}},"official":{"repos":["zhendong-wang/diffusion-policies-for-offline-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/towards-sequence-level-training-for-visual","slug":"towards-sequence-level-training-for-visual","title":"Towards Sequence-Level Training for Visual Tracking","date":"2022-08-11","arxiv_id":"2208.05810","repositories_listed":2,"syntology":{"n":5,"n_ran":3,"n_constructed":2,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/towards-sequence-level-training-for-visual#ran","syntology_url":"https://syntology.ai/paper/2208.05810","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.05810"}},"official":{"repos":["byminji/SLTtrack"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-reinforcement-learning-using-offline","slug":"robust-reinforcement-learning-using-offline","title":"Robust Reinforcement Learning using Offline Data","date":"2022-08-10","arxiv_id":"2208.05129","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":5,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 5 samples that ran constructed an object rather than computing a result","sample_list":"/paper/robust-reinforcement-learning-using-offline#ran","syntology_url":"https://syntology.ai/paper/2208.05129","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.05129"}},"official":{"repos":["zaiyan-x/RFQI"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":5,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/live-in-the-moment-learning-dynamics-model","slug":"live-in-the-moment-learning-dynamics-model","title":"Live in the Moment: Learning Dynamics Model Adapted to Evolving Policy","date":"2022-07-25","arxiv_id":"2207.12141","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/live-in-the-moment-learning-dynamics-model#ran","syntology_url":"https://syntology.ai/paper/2207.12141","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.12141"}},"official":{"repos":["si0wang/pdml"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/hierarchical-kickstarting-for-skill-transfer","slug":"hierarchical-kickstarting-for-skill-transfer","title":"Hierarchical Kickstarting for Skill Transfer in Reinforcement Learning","date":"2022-07-23","arxiv_id":"2207.11584","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hierarchical-kickstarting-for-skill-transfer#ran","syntology_url":"https://syntology.ai/paper/2207.11584","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.11584"}},"official":{"repos":["ucl-dark/skillhack"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/log-barriers-for-safe-black-box-optimization","slug":"log-barriers-for-safe-black-box-optimization","title":"Log Barriers for Safe Black-box Optimization with Application to Safe Reinforcement Learning","date":"2022-07-21","arxiv_id":"2207.10415","repositories_listed":2,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/log-barriers-for-safe-black-box-optimization#ran","syntology_url":"https://syntology.ai/paper/2207.10415","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.10415"}},"official":{"repos":["ilnura/lb_sgd","lasgroup/lbsgd-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/discriminator-weighted-offline-imitation-1","slug":"discriminator-weighted-offline-imitation-1","title":"Discriminator-Weighted Offline Imitation Learning from Suboptimal Demonstrations","date":"2022-07-20","arxiv_id":"2207.10050","repositories_listed":2,"syntology":{"n":13,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":13,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/discriminator-weighted-offline-imitation-1#ran","syntology_url":"https://syntology.ai/paper/2207.10050","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.10050"}},"official":{"repos":["ryanxhr/dwbc"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/generalizing-goal-conditioned-reinforcement","slug":"generalizing-goal-conditioned-reinforcement","title":"Generalizing Goal-Conditioned Reinforcement Learning with Variational Causal Reasoning","date":"2022-07-19","arxiv_id":"2207.09081","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generalizing-goal-conditioned-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2207.09081","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.09081"}},"official":{"repos":["gilgameshd/grader"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/active-exploration-for-inverse-reinforcement","slug":"active-exploration-for-inverse-reinforcement","title":"Active Exploration for Inverse Reinforcement Learning","date":"2022-07-18","arxiv_id":"2207.08645","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":3,"n_ran_checked":3,"n_instrument":7,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"10 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 7 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/active-exploration-for-inverse-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2207.08645","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.08645"}},"official":{"repos":["lasgroup/aceirl"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/temporal-disentanglement-of-representations","slug":"temporal-disentanglement-of-representations","title":"Temporal Disentanglement of Representations for Improved Generalisation in Reinforcement Learning","date":"2022-07-12","arxiv_id":"2207.05480","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/temporal-disentanglement-of-representations#ran","syntology_url":"https://syntology.ai/paper/2207.05480","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.05480"}},"official":{"repos":["uoe-agents/ted"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dgpo-discovering-multiple-strategies-with","slug":"dgpo-discovering-multiple-strategies-with","title":"DGPO: Discovering Multiple Strategies with Diversity-Guided Policy Optimization","date":"2022-07-12","arxiv_id":"2207.05631","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dgpo-discovering-multiple-strategies-with#ran","syntology_url":"https://syntology.ai/paper/2207.05631","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.05631"}},"official":{"repos":["OpenRL-Lab/DGPO"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/learning-bellman-complete-representations-for","slug":"learning-bellman-complete-representations-for","title":"Learning Bellman Complete Representations for Offline Policy Evaluation","date":"2022-07-12","arxiv_id":"2207.05837","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":5,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 5 samples that ran constructed an object rather than computing a result","sample_list":"/paper/learning-bellman-complete-representations-for#ran","syntology_url":"https://syntology.ai/paper/2207.05837","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.05837"}},"official":{"repos":["causalml/bcrl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":5,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/coderl-mastering-code-generation-through","slug":"coderl-mastering-code-generation-through","title":"CodeRL: Mastering Code Generation through Pretrained Models and Deep Reinforcement Learning","date":"2022-07-05","arxiv_id":"2207.01780","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/coderl-mastering-code-generation-through#ran","syntology_url":"https://syntology.ai/paper/2207.01780","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.01780"}},"official":{"repos":["salesforce/coderl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/general-policy-evaluation-and-improvement-by","slug":"general-policy-evaluation-and-improvement-by","title":"General Policy Evaluation and Improvement by Learning to Identify Few But Crucial States","date":"2022-07-04","arxiv_id":"2207.01566","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/general-policy-evaluation-and-improvement-by#ran","syntology_url":"https://syntology.ai/paper/2207.01566","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.01566"}},"official":{"repos":["idsia/policyevaluator"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/goal-conditioned-generators-of-deep-policies","slug":"goal-conditioned-generators-of-deep-policies","title":"Goal-Conditioned Generators of Deep Policies","date":"2022-07-04","arxiv_id":"2207.01570","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/goal-conditioned-generators-of-deep-policies#ran","syntology_url":"https://syntology.ai/paper/2207.01570","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.01570"}},"official":{"repos":["idsia/gogepo"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/stabilizing-off-policy-deep-reinforcement","slug":"stabilizing-off-policy-deep-reinforcement","title":"Stabilizing Off-Policy Deep Reinforcement Learning from Pixels","date":"2022-07-03","arxiv_id":"2207.00986","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/stabilizing-off-policy-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2207.00986","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.00986"}},"official":{"repos":["aladoro/stabilizing-off-policy-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/usher-unbiased-sampling-for-hindsight","slug":"usher-unbiased-sampling-for-hindsight","title":"USHER: Unbiased Sampling for Hindsight Experience Replay","date":"2022-07-03","arxiv_id":"2207.01115","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/usher-unbiased-sampling-for-hindsight#ran","syntology_url":"https://syntology.ai/paper/2207.01115","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.01115"}},"official":null}},{"url":"/paper/modular-lifelong-reinforcement-learning-via-1","slug":"modular-lifelong-reinforcement-learning-via-1","title":"Modular Lifelong Reinforcement Learning via Neural Composition","date":"2022-07-01","arxiv_id":"2207.00429","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/modular-lifelong-reinforcement-learning-via-1#ran","syntology_url":"https://syntology.ai/paper/2207.00429","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.00429"}},"official":{"repos":["lifelong-ml/mendez2022modularlifelongrl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-the-learning-and-learnablity-of","slug":"on-the-learning-and-learnablity-of","title":"On the Learning and Learnability of Quasimetrics","date":"2022-06-30","arxiv_id":"2206.15478","repositories_listed":2,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/on-the-learning-and-learnablity-of#ran","syntology_url":"https://syntology.ai/paper/2206.15478","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.15478"}},"official":{"repos":["ssnl/poisson_quasimetric_embedding"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/distinguishing-learning-rules-with-brain","slug":"distinguishing-learning-rules-with-brain","title":"Distinguishing Learning Rules with Brain Machine Interfaces","date":"2022-06-27","arxiv_id":"2206.13448","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/distinguishing-learning-rules-with-brain#ran","syntology_url":"https://syntology.ai/paper/2206.13448","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.13448"}},"official":{"repos":["jacobfulano/learning-rules-with-bmi"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/video-pretraining-vpt-learning-to-act-by","slug":"video-pretraining-vpt-learning-to-act-by","title":"Video PreTraining (VPT): Learning to Act by Watching Unlabeled Online Videos","date":"2022-06-23","arxiv_id":"2206.11795","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/video-pretraining-vpt-learning-to-act-by#ran","syntology_url":"https://syntology.ai/paper/2206.11795","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.11795"}},"official":{"repos":["openai/Video-Pre-Training"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/optimistic-linear-support-and-successor","slug":"optimistic-linear-support-and-successor","title":"Optimistic Linear Support and Successor Features as a Basis for Optimal Policy Transfer","date":"2022-06-22","arxiv_id":"2206.11326","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":1,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/optimistic-linear-support-and-successor#ran","syntology_url":"https://syntology.ai/paper/2206.11326","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.11326"}},"official":{"repos":["lucasalegre/sfols"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pac-assisted-value-factorisation-with","slug":"pac-assisted-value-factorisation-with","title":"PAC: Assisted Value Factorisation with Counterfactual Predictions in Multi-Agent Reinforcement Learning","date":"2022-06-22","arxiv_id":"2206.11420","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/pac-assisted-value-factorisation-with#ran","syntology_url":"https://syntology.ai/paper/2206.11420","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.11420"}},"official":{"repos":["hanhananderson/pac-marl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-deep-reinforcement-learning-through-1","slug":"robust-deep-reinforcement-learning-through-1","title":"Robust Deep Reinforcement Learning through Bootstrapped Opportunistic Curriculum","date":"2022-06-21","arxiv_id":"2206.10057","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-deep-reinforcement-learning-through-1#ran","syntology_url":"https://syntology.ai/paper/2206.10057","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.10057"}},"official":{"repos":["jlwu002/bcl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/envpool-a-highly-parallel-reinforcement","slug":"envpool-a-highly-parallel-reinforcement","title":"EnvPool: A Highly Parallel Reinforcement Learning Environment Execution Engine","date":"2022-06-21","arxiv_id":"2206.10558","repositories_listed":3,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/envpool-a-highly-parallel-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2206.10558","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.10558"}},"official":{"repos":["sail-sg/envpool"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["named_in_paper","official"]}}},{"url":"/paper/benchmarking-constraint-inference-in-inverse","slug":"benchmarking-constraint-inference-in-inverse","title":"Benchmarking Constraint Inference in Inverse Reinforcement Learning","date":"2022-06-20","arxiv_id":"2206.09670","repositories_listed":2,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-constraint-inference-in-inverse#ran","syntology_url":"https://syntology.ai/paper/2206.09670","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.09670"}},"official":{"repos":["guiliang/cirl-benchmarks-public","guiliang/icrl-benchmarks-public"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bootstrapped-transformer-for-offline","slug":"bootstrapped-transformer-for-offline","title":"Bootstrapped Transformer for Offline Reinforcement Learning","date":"2022-06-17","arxiv_id":"2206.08569","repositories_listed":0,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bootstrapped-transformer-for-offline#ran","syntology_url":"https://syntology.ai/paper/2206.08569","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.08569"}},"official":null}},{"url":"/paper/smpl-simulated-industrial-manufacturing-and","slug":"smpl-simulated-industrial-manufacturing-and","title":"SMPL: Simulated Industrial Manufacturing and Process Control Learning Environments","date":"2022-06-17","arxiv_id":"2206.08851","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/smpl-simulated-industrial-manufacturing-and#ran","syntology_url":"https://syntology.ai/paper/2206.08851","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.08851"}},"official":{"repos":["smpl-env/smpl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fast-population-based-reinforcement-learning","slug":"fast-population-based-reinforcement-learning","title":"Fast Population-Based Reinforcement Learning on a Single Machine","date":"2022-06-17","arxiv_id":"2206.08888","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fast-population-based-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2206.08888","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.08888"}},"official":null}},{"url":"/paper/double-check-your-state-before-trusting-it","slug":"double-check-your-state-before-trusting-it","title":"Double Check Your State Before Trusting It: Confidence-Aware Bidirectional Offline Model-Based Imagination","date":"2022-06-16","arxiv_id":"2206.07989","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":1,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/double-check-your-state-before-trusting-it#ran","syntology_url":"https://syntology.ai/paper/2206.07989","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.07989"}},"official":{"repos":["dmksjfl/CABI"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/training-discrete-deep-generative-models-via","slug":"training-discrete-deep-generative-models-via","title":"Training Discrete Deep Generative Models via Gapped Straight-Through Estimator","date":"2022-06-15","arxiv_id":"2206.07235","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/training-discrete-deep-generative-models-via#ran","syntology_url":"https://syntology.ai/paper/2206.07235","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.07235"}},"official":{"repos":["chijames/gst"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/transformers-are-meta-reinforcement-learners-1","slug":"transformers-are-meta-reinforcement-learners-1","title":"Transformers are Meta-Reinforcement Learners","date":"2022-06-14","arxiv_id":"2206.06614","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/transformers-are-meta-reinforcement-learners-1#ran","syntology_url":"https://syntology.ai/paper/2206.06614","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.06614"}},"official":{"repos":["luckeciano/transformers-metarl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/regularizing-a-model-based-policy-stationary","slug":"regularizing-a-model-based-policy-stationary","title":"Regularizing a Model-based Policy Stationary Distribution to Stabilize Offline Reinforcement Learning","date":"2022-06-14","arxiv_id":"2206.07166","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/regularizing-a-model-based-policy-stationary#ran","syntology_url":"https://syntology.ai/paper/2206.07166","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.07166"}},"official":{"repos":["shentao-yang/sdm-gan_icml2022"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/defending-observation-attacks-in-deep","slug":"defending-observation-attacks-in-deep","title":"Defending Observation Attacks in Deep Reinforcement Learning via Detection and Denoising","date":"2022-06-14","arxiv_id":"2206.07188","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/defending-observation-attacks-in-deep#ran","syntology_url":"https://syntology.ai/paper/2206.07188","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.07188"}},"official":{"repos":["ZikangXiong/rl-detect-and-denoise-defense"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-unified-approach-to-reinforcement-learning","slug":"a-unified-approach-to-reinforcement-learning","title":"A Unified Approach to Reinforcement Learning, Quantal Response Equilibria, and Two-Player Zero-Sum Games","date":"2022-06-12","arxiv_id":"2206.05825","repositories_listed":3,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":3,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 3 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-unified-approach-to-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2206.05825","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.05825"}},"official":{"repos":["deepmind/open_spiel"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["named_in_paper"]}}},{"url":"/paper/does-self-supervised-learning-really-improve","slug":"does-self-supervised-learning-really-improve","title":"Does Self-supervised Learning Really Improve Reinforcement Learning from Pixels?","date":"2022-06-10","arxiv_id":"2206.05266","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":3,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/does-self-supervised-learning-really-improve#ran","syntology_url":"https://syntology.ai/paper/2206.05266","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.05266"}},"official":{"repos":["LostXine/elo-sac","lostxine/elo-rainbow"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/anchor-changing-regularized-natural-policy","slug":"anchor-changing-regularized-natural-policy","title":"Anchor-Changing Regularized Natural Policy Gradient for Multi-Objective Reinforcement Learning","date":"2022-06-10","arxiv_id":"2206.05357","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/anchor-changing-regularized-natural-policy#ran","syntology_url":"https://syntology.ai/paper/2206.05357","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.05357"}},"official":{"repos":["tliu1997/arnpg-morl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mildly-conservative-q-learning-for-offline","slug":"mildly-conservative-q-learning-for-offline","title":"Mildly Conservative Q-Learning for Offline Reinforcement Learning","date":"2022-06-09","arxiv_id":"2206.04745","repositories_listed":3,"syntology":{"n":16,"n_ran":13,"n_constructed":6,"n_ran_checked":13,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":12,"n_pointer_only":11,"phrase":"13 ran (of which 6 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 1 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/mildly-conservative-q-learning-for-offline#ran","syntology_url":"https://syntology.ai/paper/2206.04745","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.04745"}},"official":{"repos":["dmksjfl/mcq"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text","listed"]}}},{"url":"/paper/how-far-i-ll-go-offline-goal-conditioned","slug":"how-far-i-ll-go-offline-goal-conditioned","title":"How Far I'll Go: Offline Goal-Conditioned Reinforcement Learning via $f$-Advantage Regression","date":"2022-06-07","arxiv_id":"2206.03023","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/how-far-i-ll-go-offline-goal-conditioned#ran","syntology_url":"https://syntology.ai/paper/2206.03023","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.03023"}},"official":{"repos":["jasonma2016/gofar"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rorl-robust-offline-reinforcement-learning","slug":"rorl-robust-offline-reinforcement-learning","title":"RORL: Robust Offline Reinforcement Learning via Conservative Smoothing","date":"2022-06-06","arxiv_id":"2206.02829","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rorl-robust-offline-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2206.02829","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.02829"}},"official":{"repos":["yangrui2015/rorl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-tabula-rasa-reincarnating","slug":"beyond-tabula-rasa-reincarnating","title":"Reincarnating Reinforcement Learning: Reusing Prior Computation to Accelerate Progress","date":"2022-06-03","arxiv_id":"2206.01626","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/beyond-tabula-rasa-reincarnating#ran","syntology_url":"https://syntology.ai/paper/2206.01626","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.01626"}},"official":{"repos":["google-research/reincarnating_rl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-transformer-q-networks-for-partially","slug":"deep-transformer-q-networks-for-partially","title":"Deep Transformer Q-Networks for Partially Observable Reinforcement Learning","date":"2022-06-02","arxiv_id":"2206.01078","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-transformer-q-networks-for-partially#ran","syntology_url":"https://syntology.ai/paper/2206.01078","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.01078"}},"official":{"repos":["kevslinger/dtqn"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/when-does-return-conditioned-supervised","slug":"when-does-return-conditioned-supervised","title":"When does return-conditioned supervised learning work for offline reinforcement learning?","date":"2022-06-02","arxiv_id":"2206.01079","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/when-does-return-conditioned-supervised#ran","syntology_url":"https://syntology.ai/paper/2206.01079","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.01079"}},"official":{"repos":["davidbrandfonbrener/rcsl-paper"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/on-reinforcement-learning-and-distribution","slug":"on-reinforcement-learning-and-distribution","title":"On Reinforcement Learning and Distribution Matching for Fine-Tuning Language Models with no Catastrophic Forgetting","date":"2022-06-01","arxiv_id":"2206.00761","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/on-reinforcement-learning-and-distribution#ran","syntology_url":"https://syntology.ai/paper/2206.00761","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.00761"}},"official":{"repos":["naver/gdc"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-reward-poisoning-attacks-on-online","slug":"efficient-reward-poisoning-attacks-on-online","title":"Efficient Reward Poisoning Attacks on Online Deep Reinforcement Learning","date":"2022-05-30","arxiv_id":"2205.14842","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-reward-poisoning-attacks-on-online#ran","syntology_url":"https://syntology.ai/paper/2205.14842","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.14842"}},"official":{"repos":["yinglunxu/reward_poisoning_attack_drl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-with-a-terminator","slug":"reinforcement-learning-with-a-terminator","title":"Reinforcement Learning with a Terminator","date":"2022-05-30","arxiv_id":"2205.15376","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-with-a-terminator#ran","syntology_url":"https://syntology.ai/paper/2205.15376","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.15376"}},"official":{"repos":["guytenn/terminator"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dep-rl-embodied-exploration-for-reinforcement","slug":"dep-rl-embodied-exploration-for-reinforcement","title":"DEP-RL: Embodied Exploration for Reinforcement Learning in Overactuated and Musculoskeletal Systems","date":"2022-05-30","arxiv_id":"2206.00484","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/dep-rl-embodied-exploration-for-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2206.00484","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.00484"}},"official":{"repos":["martius-lab/depRL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/on-the-robustness-of-safe-reinforcement","slug":"on-the-robustness-of-safe-reinforcement","title":"On the Robustness of Safe Reinforcement Learning under Observational Perturbations","date":"2022-05-29","arxiv_id":"2205.14691","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":6,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 1 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/on-the-robustness-of-safe-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2205.14691","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.14691"}},"official":{"repos":["liuzuxin/safe-rl-robustness"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/task-agnostic-continual-reinforcement","slug":"task-agnostic-continual-reinforcement","title":"Task-Agnostic Continual Reinforcement Learning: Gaining Insights and Overcoming Challenges","date":"2022-05-28","arxiv_id":"2205.14495","repositories_listed":2,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/task-agnostic-continual-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2205.14495","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.14495"}},"official":{"repos":["amazon-science/replay-based-recurrent-rl","amazon-research/replay-based-recurrent-rl"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/fedformer-contextual-federation-with","slug":"fedformer-contextual-federation-with","title":"FedFormer: Contextual Federation with Attention in Reinforcement Learning","date":"2022-05-27","arxiv_id":"2205.13697","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fedformer-contextual-federation-with#ran","syntology_url":"https://syntology.ai/paper/2205.13697","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.13697"}},"official":{"repos":["liamhebert/FedFormer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-solve-combinatorial-graph","slug":"learning-to-solve-combinatorial-graph","title":"Learning to Solve Combinatorial Graph Partitioning Problems via Efficient Exploration","date":"2022-05-27","arxiv_id":"2205.14105","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/learning-to-solve-combinatorial-graph#ran","syntology_url":"https://syntology.ai/paper/2205.14105","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.14105"}},"official":{"repos":["tomdbar/ecord"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/tiered-reinforcement-learning-pessimism-in","slug":"tiered-reinforcement-learning-pessimism-in","title":"Tiered Reinforcement Learning: Pessimism in the Face of Uncertainty and Constant Regret","date":"2022-05-25","arxiv_id":"2205.12418","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/tiered-reinforcement-learning-pessimism-in#ran","syntology_url":"https://syntology.ai/paper/2205.12418","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.12418"}},"official":{"repos":["jiaweihhuang/tiered-rl-experiments"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/skill-machines-temporal-logic-composition-in","slug":"skill-machines-temporal-logic-composition-in","title":"Skill Machines: Temporal Logic Skill Composition in Reinforcement Learning","date":"2022-05-25","arxiv_id":"2205.12532","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":6,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/skill-machines-temporal-logic-composition-in#ran","syntology_url":"https://syntology.ai/paper/2205.12532","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.12532"}},"official":{"repos":["geraudnt/skill_machines"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/history-compression-via-language-models-in","slug":"history-compression-via-language-models-in","title":"History Compression via Language Models in Reinforcement Learning","date":"2022-05-24","arxiv_id":"2205.12258","repositories_listed":2,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/history-compression-via-language-models-in#ran","syntology_url":"https://syntology.ai/paper/2205.12258","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.12258"}},"official":{"repos":["ml-jku/helm"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/reward-uncertainty-for-exploration-in-1","slug":"reward-uncertainty-for-exploration-in-1","title":"Reward Uncertainty for Exploration in Preference-based Reinforcement Learning","date":"2022-05-24","arxiv_id":"2205.12401","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":2,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reward-uncertainty-for-exploration-in-1#ran","syntology_url":"https://syntology.ai/paper/2205.12401","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.12401"}},"official":null}},{"url":"/paper/learning-to-branch-with-tree-mdps","slug":"learning-to-branch-with-tree-mdps","title":"Learning to branch with Tree MDPs","date":"2022-05-23","arxiv_id":"2205.11107","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-to-branch-with-tree-mdps#ran","syntology_url":"https://syntology.ai/paper/2205.11107","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.11107"}},"official":{"repos":["lascavana/rl2branch"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/an-evaluation-study-of-intrinsic-motivation","slug":"an-evaluation-study-of-intrinsic-motivation","title":"An Evaluation Study of Intrinsic Motivation Techniques applied to Reinforcement Learning over Hard Exploration Environments","date":"2022-05-23","arxiv_id":"2205.11184","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-evaluation-study-of-intrinsic-motivation#ran","syntology_url":"https://syntology.ai/paper/2205.11184","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.11184"}},"official":{"repos":["aklein1995/intrinsic_motivation_techniques_study"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/memory-efficient-reinforcement-learning-with","slug":"memory-efficient-reinforcement-learning-with","title":"Memory-efficient Reinforcement Learning with Value-based Knowledge Consolidation","date":"2022-05-22","arxiv_id":"2205.10868","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/memory-efficient-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2205.10868","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.10868"}},"official":{"repos":["qlan3/MeDQN"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/reachability-constrained-reinforcement","slug":"reachability-constrained-reinforcement","title":"Reachability Constrained Reinforcement Learning","date":"2022-05-16","arxiv_id":"2205.07536","repositories_listed":2,"syntology":{"n":5,"n_ran":4,"n_constructed":3,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/reachability-constrained-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2205.07536","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.07536"}},"official":{"repos":["mahaitongdae/Reachability_Constrained_RL"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/cliff-diving-exploring-reward-surfaces-in","slug":"cliff-diving-exploring-reward-surfaces-in","title":"Cliff Diving: Exploring Reward Surfaces in Reinforcement Learning Environments","date":"2022-05-14","arxiv_id":"2205.07015","repositories_listed":0,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/cliff-diving-exploring-reward-surfaces-in#ran","syntology_url":"https://syntology.ai/paper/2205.07015","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.07015"}},"official":null}},{"url":"/paper/a-state-distribution-matching-approach-to-non","slug":"a-state-distribution-matching-approach-to-non","title":"A State-Distribution Matching Approach to Non-Episodic Reinforcement Learning","date":"2022-05-11","arxiv_id":"2205.05212","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-state-distribution-matching-approach-to-non#ran","syntology_url":"https://syntology.ai/paper/2205.05212","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.05212"}},"official":{"repos":["architsharma97/medal"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-risk-averse-reinforcement-learning","slug":"efficient-risk-averse-reinforcement-learning","title":"Efficient Risk-Averse Reinforcement Learning","date":"2022-05-10","arxiv_id":"2205.05138","repositories_listed":2,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":2,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/efficient-risk-averse-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2205.05138","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.05138"}},"official":{"repos":["ido90/CeSoR"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/multivariate-prediction-intervals-for-random","slug":"multivariate-prediction-intervals-for-random","title":"Multivariate Prediction Intervals for Random Forests","date":"2022-05-04","arxiv_id":"2205.02260","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multivariate-prediction-intervals-for-random#ran","syntology_url":"https://syntology.ai/paper/2205.02260","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.02260"}},"official":{"repos":["CitrineInformatics/lolo","citrineinformatics/multivariate-prediction-intervals"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/cclf-a-contrastive-curiosity-driven-learning","slug":"cclf-a-contrastive-curiosity-driven-learning","title":"CCLF: A Contrastive-Curiosity-Driven Learning Framework for Sample-Efficient Reinforcement Learning","date":"2022-05-02","arxiv_id":"2205.00943","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/cclf-a-contrastive-curiosity-driven-learning#ran","syntology_url":"https://syntology.ai/paper/2205.00943","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.00943"}},"official":{"repos":["csun001/cclf"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/ttopt-a-maximum-volume-quantized-tensor-train","slug":"ttopt-a-maximum-volume-quantized-tensor-train","title":"TTOpt: A Maximum Volume Quantized Tensor Train-based Optimization and its Application to Reinforcement Learning","date":"2022-04-30","arxiv_id":"2205.00293","repositories_listed":1,"syntology":{"n":17,"n_ran":14,"n_constructed":0,"n_ran_checked":6,"n_instrument":8,"n_unverified":3,"n_honours":6,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 6 honoured, 0 violated, 0 with no contract checked; 8 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/ttopt-a-maximum-volume-quantized-tensor-train#ran","syntology_url":"https://syntology.ai/paper/2205.00293","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.00293"}},"official":{"repos":["andreichertkov/ttopt"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/markov-abstractions-for-pac-reinforcement","slug":"markov-abstractions-for-pac-reinforcement","title":"Markov Abstractions for PAC Reinforcement Learning in Non-Markov Decision Processes","date":"2022-04-29","arxiv_id":"2205.01053","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/markov-abstractions-for-pac-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2205.01053","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.01053"}},"official":{"repos":["whitemech/markov-abstractions-code-ijcai22"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rambo-rl-robust-adversarial-model-based","slug":"rambo-rl-robust-adversarial-model-based","title":"RAMBO-RL: Robust Adversarial Model-Based Offline Reinforcement Learning","date":"2022-04-26","arxiv_id":"2204.12581","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rambo-rl-robust-adversarial-model-based#ran","syntology_url":"https://syntology.ai/paper/2204.12581","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.12581"}},"official":{"repos":["marc-rigter/rambo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hypernca-growing-developmental-networks-with","slug":"hypernca-growing-developmental-networks-with","title":"HyperNCA: Growing Developmental Networks with Neural Cellular Automata","date":"2022-04-25","arxiv_id":"2204.11674","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/hypernca-growing-developmental-networks-with#ran","syntology_url":"https://syntology.ai/paper/2204.11674","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.11674"}},"official":null}},{"url":"/paper/multi-objective-pointer-network-for","slug":"multi-objective-pointer-network-for","title":"Multi-objective Pointer Network for Combinatorial Optimization","date":"2022-04-25","arxiv_id":"2204.11860","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multi-objective-pointer-network-for#ran","syntology_url":"https://syntology.ai/paper/2204.11860","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.11860"}},"official":{"repos":["gaoly/mopn"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/coptidice-offline-constrained-reinforcement-1","slug":"coptidice-offline-constrained-reinforcement-1","title":"COptiDICE: Offline Constrained Reinforcement Learning via Stationary Distribution Correction Estimation","date":"2022-04-19","arxiv_id":"2204.08957","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/coptidice-offline-constrained-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/2204.08957","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.08957"}},"official":{"repos":["deepmind/constrained_optidice"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/chai-a-chatbot-ai-for-task-oriented-dialogue","slug":"chai-a-chatbot-ai-for-task-oriented-dialogue","title":"CHAI: A CHatbot AI for Task-Oriented Dialogue with Offline Reinforcement Learning","date":"2022-04-18","arxiv_id":"2204.08426","repositories_listed":2,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/chai-a-chatbot-ai-for-task-oriented-dialogue#ran","syntology_url":"https://syntology.ai/paper/2204.08426","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.08426"}},"official":{"repos":["siddharthverma314/chai-naacl-2022"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/can-question-rewriting-help-conversational","slug":"can-question-rewriting-help-conversational","title":"Can Question Rewriting Help Conversational Question Answering?","date":"2022-04-13","arxiv_id":"2204.06239","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/can-question-rewriting-help-conversational#ran","syntology_url":"https://syntology.ai/paper/2204.06239","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.06239"}},"official":{"repos":["hltchkust/cqr4cqa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/training-a-helpful-and-harmless-assistant","slug":"training-a-helpful-and-harmless-assistant","title":"Training a Helpful and Harmless Assistant with Reinforcement Learning from Human Feedback","date":"2022-04-12","arxiv_id":"2204.05862","repositories_listed":4,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/training-a-helpful-and-harmless-assistant#ran","syntology_url":"https://syntology.ai/paper/2204.05862","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.05862"}},"official":{"repos":["anthropics/hh-rlhf"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/grounding-hindsight-instructions-in-multi","slug":"grounding-hindsight-instructions-in-multi","title":"Grounding Hindsight Instructions in Multi-Goal Reinforcement Learning for Robotics","date":"2022-04-08","arxiv_id":"2204.04308","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/grounding-hindsight-instructions-in-multi#ran","syntology_url":"https://syntology.ai/paper/2204.04308","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.04308"}},"official":{"repos":["knowledgetechnologyuhh/hipss"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/asynchronous-reinforcement-learning-for-real","slug":"asynchronous-reinforcement-learning-for-real","title":"Asynchronous Reinforcement Learning for Real-Time Control of Physical Robots","date":"2022-03-23","arxiv_id":"2203.12759","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":2,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/asynchronous-reinforcement-learning-for-real#ran","syntology_url":"https://syntology.ai/paper/2203.12759","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.12759"}},"official":{"repos":["yufengyuan/ur5_async_rl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/insights-from-the-neurips-2021-nethack","slug":"insights-from-the-neurips-2021-nethack","title":"Insights From the NeurIPS 2021 NetHack Challenge","date":"2022-03-22","arxiv_id":"2203.11889","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/insights-from-the-neurips-2021-nethack#ran","syntology_url":"https://syntology.ai/paper/2203.11889","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.11889"}},"official":null}},{"url":"/paper/teachable-reinforcement-learning-via-advice-1","slug":"teachable-reinforcement-learning-via-advice-1","title":"Teachable Reinforcement Learning via Advice Distillation","date":"2022-03-19","arxiv_id":"2203.11197","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/teachable-reinforcement-learning-via-advice-1#ran","syntology_url":"https://syntology.ai/paper/2203.11197","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.11197"}},"official":{"repos":["rll-research/teachable"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/pmic-improving-multi-agent-reinforcement-1","slug":"pmic-improving-multi-agent-reinforcement-1","title":"PMIC: Improving Multi-Agent Reinforcement Learning with Progressive Mutual Information Collaboration","date":"2022-03-16","arxiv_id":"2203.08553","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pmic-improving-multi-agent-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/2203.08553","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.08553"}},"official":{"repos":["yeshenpy/pmic"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/latent-variable-advantage-weighted-policy","slug":"latent-variable-advantage-weighted-policy","title":"Latent-Variable Advantage-Weighted Policy Optimization for Offline RL","date":"2022-03-16","arxiv_id":"2203.08949","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/latent-variable-advantage-weighted-policy#ran","syntology_url":"https://syntology.ai/paper/2203.08949","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.08949"}},"official":null}},{"url":"/paper/zipfian-environments-for-reinforcement","slug":"zipfian-environments-for-reinforcement","title":"Zipfian environments for Reinforcement Learning","date":"2022-03-15","arxiv_id":"2203.08222","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/zipfian-environments-for-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2203.08222","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.08222"}},"official":{"repos":["deepmind/zipfian_environments"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-for-entity-1","slug":"deep-reinforcement-learning-for-entity-1","title":"Deep Reinforcement Learning for Entity Alignment","date":"2022-03-07","arxiv_id":"2203.03315","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":4,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/deep-reinforcement-learning-for-entity-1#ran","syntology_url":"https://syntology.ai/paper/2203.03315","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.03315"}},"official":{"repos":["guolingbing/rlea"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/influencing-long-term-behavior-in-multiagent","slug":"influencing-long-term-behavior-in-multiagent","title":"Influencing Long-Term Behavior in Multiagent Reinforcement Learning","date":"2022-03-07","arxiv_id":"2203.03535","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":3,"n_ran_checked":4,"n_instrument":6,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"10 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 6 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/influencing-long-term-behavior-in-multiagent#ran","syntology_url":"https://syntology.ai/paper/2203.03535","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.03535"}},"official":{"repos":["dkkim93/further"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/a-survey-on-offline-reinforcement-learning","slug":"a-survey-on-offline-reinforcement-learning","title":"A Survey on Offline Reinforcement Learning: Taxonomy, Review, and Open Problems","date":"2022-03-02","arxiv_id":"2203.01387","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-survey-on-offline-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2203.01387","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.01387"}},"official":{"repos":["larocs/offline-rl-suvey"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/quantum-deep-reinforcement-learning-for-robot","slug":"quantum-deep-reinforcement-learning-for-robot","title":"Quantum Deep Reinforcement Learning for Robot Navigation Tasks","date":"2022-02-24","arxiv_id":"2202.12180","repositories_listed":2,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/quantum-deep-reinforcement-learning-for-robot#ran","syntology_url":"https://syntology.ai/paper/2202.12180","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.12180"}},"official":{"repos":["dfki-ric-quantum/qdrl-turtlebot-env","dfki-ric-quantum/qdrl-turtlebot-eval"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-transferable-reward-for-query-object-1","slug":"learning-transferable-reward-for-query-object-1","title":"Learning Transferable Reward for Query Object Localization with Policy Adaptation","date":"2022-02-24","arxiv_id":"2202.12403","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-transferable-reward-for-query-object-1#ran","syntology_url":"https://syntology.ai/paper/2202.12403","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.12403"}},"official":{"repos":["litingfeng/localization-by-ordembed"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/transdreamer-reinforcement-learning-with-1","slug":"transdreamer-reinforcement-learning-with-1","title":"TransDreamer: Reinforcement Learning with Transformer World Models","date":"2022-02-19","arxiv_id":"2202.09481","repositories_listed":0,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/transdreamer-reinforcement-learning-with-1#ran","syntology_url":"https://syntology.ai/paper/2202.09481","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.09481"}},"official":null}},{"url":"/paper/cadre-a-cascade-deep-reinforcement-learning","slug":"cadre-a-cascade-deep-reinforcement-learning","title":"CADRE: A Cascade Deep Reinforcement Learning Framework for Vision-based Autonomous Urban Driving","date":"2022-02-17","arxiv_id":"2202.08557","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":7,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 7 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cadre-a-cascade-deep-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2202.08557","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.08557"}},"official":{"repos":["BIT-MCS/Cadre"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":7,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/vrl3-a-data-driven-framework-for-visual-deep","slug":"vrl3-a-data-driven-framework-for-visual-deep","title":"VRL3: A Data-Driven Framework for Visual Deep Reinforcement Learning","date":"2022-02-17","arxiv_id":"2202.10324","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":4,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/vrl3-a-data-driven-framework-for-visual-deep#ran","syntology_url":"https://syntology.ai/paper/2202.10324","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.10324"}},"official":{"repos":["facebookresearch/drqv2"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/safe-reinforcement-learning-by-imagining-the-1","slug":"safe-reinforcement-learning-by-imagining-the-1","title":"Safe Reinforcement Learning by Imagining the Near Future","date":"2022-02-15","arxiv_id":"2202.07789","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/safe-reinforcement-learning-by-imagining-the-1#ran","syntology_url":"https://syntology.ai/paper/2202.07789","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.07789"}},"official":{"repos":["gwthomas/safe-mbpo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}}],"record_sha256":"e518ce94bf76b9bb5b408e8f1942be725e157c852ba6d975f936c46ba7a14b97","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}