{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/ran/5","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":5,"pages_in_order":15,"rows_per_page":100,"rows":[401,500],"of":1416,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1/papers/ran/1","prev":"/task/reinforcement-learning-1/papers/ran/4","next":"/task/reinforcement-learning-1/papers/ran/6","papers":[{"url":"/paper/hiql-offline-goal-conditioned-rl-with-latent-1","slug":"hiql-offline-goal-conditioned-rl-with-latent-1","title":"HIQL: Offline Goal-Conditioned RL with Latent States as Actions","date":"2023-07-22","arxiv_id":"2307.11949","repositories_listed":1,"syntology":{"n":17,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":9,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/hiql-offline-goal-conditioned-rl-with-latent-1#ran","syntology_url":"https://syntology.ai/paper/2307.11949","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.11949"}},"official":{"repos":["seohongpark/hiql"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/offline-multi-agent-reinforcement-learning-1","slug":"offline-multi-agent-reinforcement-learning-1","title":"Offline Multi-Agent Reinforcement Learning with Implicit Global-to-Local Value Regularization","date":"2023-07-21","arxiv_id":"2307.11620","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":4,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","sample_list":"/paper/offline-multi-agent-reinforcement-learning-1#ran","syntology_url":"https://syntology.ai/paper/2307.11620","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.11620"}},"official":{"repos":["zhengyinan-air/omiga"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/natural-actor-critic-for-robust-reinforcement","slug":"natural-actor-critic-for-robust-reinforcement","title":"Natural Actor-Critic for Robust Reinforcement Learning with Function Approximation","date":"2023-07-17","arxiv_id":"2307.08875","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":8,"n_ran_checked":9,"n_instrument":2,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":14,"phrase":"11 ran (of which 8 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/natural-actor-critic-for-robust-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2307.08875","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.08875"}},"official":{"repos":["tliu1997/rnac"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":8,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/safe-dreamerv3-safe-reinforcement-learning","slug":"safe-dreamerv3-safe-reinforcement-learning","title":"SafeDreamer: Safe Reinforcement Learning with World Models","date":"2023-07-14","arxiv_id":"2307.07176","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":4,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/safe-dreamerv3-safe-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2307.07176","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.07176"}},"official":{"repos":["pku-alignment/safedreamer"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/robotic-manipulation-datasets-for-offline","slug":"robotic-manipulation-datasets-for-offline","title":"Robotic Manipulation Datasets for Offline Compositional Reinforcement Learning","date":"2023-07-13","arxiv_id":"2307.07091","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robotic-manipulation-datasets-for-offline#ran","syntology_url":"https://syntology.ai/paper/2307.07091","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.07091"}},"official":{"repos":["lifelong-ml/offline-compositional-rl-datasets"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pid-inspired-inductive-biases-for-deep-1","slug":"pid-inspired-inductive-biases-for-deep-1","title":"PID-Inspired Inductive Biases for Deep Reinforcement Learning in Partially Observable Control Tasks","date":"2023-07-12","arxiv_id":"2307.05891","repositories_listed":2,"syntology":{"n":9,"n_ran":8,"n_constructed":1,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"8 ran (of which 1 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pid-inspired-inductive-biases-for-deep-1#ran","syntology_url":"https://syntology.ai/paper/2307.05891","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.05891"}},"official":null}},{"url":"/paper/rltf-reinforcement-learning-from-unit-test","slug":"rltf-reinforcement-learning-from-unit-test","title":"RLTF: Reinforcement Learning from Unit Test Feedback","date":"2023-07-10","arxiv_id":"2307.04349","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/rltf-reinforcement-learning-from-unit-test#ran","syntology_url":"https://syntology.ai/paper/2307.04349","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.04349"}},"official":{"repos":["zyq-scut/rltf"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/alleviating-matthew-effect-of-offline","slug":"alleviating-matthew-effect-of-offline","title":"Alleviating Matthew Effect of Offline Reinforcement Learning in Interactive Recommendation","date":"2023-07-10","arxiv_id":"2307.04571","repositories_listed":2,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/alleviating-matthew-effect-of-offline#ran","syntology_url":"https://syntology.ai/paper/2307.04571","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.04571"}},"official":{"repos":["chongminggao/dorl-codes"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/when-do-transformers-shine-in-rl-decoupling-1","slug":"when-do-transformers-shine-in-rl-decoupling-1","title":"When Do Transformers Shine in RL? Decoupling Memory from Credit Assignment","date":"2023-07-07","arxiv_id":"2307.03864","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/when-do-transformers-shine-in-rl-decoupling-1#ran","syntology_url":"https://syntology.ai/paper/2307.03864","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.03864"}},"official":{"repos":["twni2016/memory-rl","twni2016/pomdp-baselines"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/first-explore-then-exploit-meta-learning","slug":"first-explore-then-exploit-meta-learning","title":"First-Explore, then Exploit: Meta-Learning to Solve Hard Exploration-Exploitation Trade-Offs","date":"2023-07-05","arxiv_id":"2307.02276","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/first-explore-then-exploit-meta-learning#ran","syntology_url":"https://syntology.ai/paper/2307.02276","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.02276"}},"official":{"repos":["btnorman/First-Explore"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/rl4co-an-extensive-reinforcement-learning-for","slug":"rl4co-an-extensive-reinforcement-learning-for","title":"RL4CO: an Extensive Reinforcement Learning for Combinatorial Optimization Benchmark","date":"2023-06-29","arxiv_id":"2306.17100","repositories_listed":3,"syntology":{"n":16,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":3,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/rl4co-an-extensive-reinforcement-learning-for#ran","syntology_url":"https://syntology.ai/paper/2306.17100","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.17100"}},"official":{"repos":["ai4co/rl4co","pytorch/rl"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/learning-to-modulate-pre-trained-models-in-rl-1","slug":"learning-to-modulate-pre-trained-models-in-rl-1","title":"Learning to Modulate pre-trained Models in RL","date":"2023-06-26","arxiv_id":"2306.14884","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-to-modulate-pre-trained-models-in-rl-1#ran","syntology_url":"https://syntology.ai/paper/2306.14884","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.14884"}},"official":{"repos":["ml-jku/l2m"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/intercode-standardizing-and-benchmarking","slug":"intercode-standardizing-and-benchmarking","title":"InterCode: Standardizing and Benchmarking Interactive Coding with Execution Feedback","date":"2023-06-26","arxiv_id":"2306.14898","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/intercode-standardizing-and-benchmarking#ran","syntology_url":"https://syntology.ai/paper/2306.14898","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.14898"}},"official":{"repos":["princeton-nlp/intercode"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-ood-state-actions-supported-cross","slug":"beyond-ood-state-actions-supported-cross","title":"Beyond OOD State Actions: Supported Cross-Domain Offline Reinforcement Learning","date":"2023-06-22","arxiv_id":"2306.12755","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/beyond-ood-state-actions-supported-cross#ran","syntology_url":"https://syntology.ai/paper/2306.12755","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.12755"}},"official":{"repos":["thuml/SPOT"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/taco-temporal-latent-action-driven","slug":"taco-temporal-latent-action-driven","title":"TACO: Temporal Latent Action-Driven Contrastive Loss for Visual Reinforcement Learning","date":"2023-06-22","arxiv_id":"2306.13229","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/taco-temporal-latent-action-driven#ran","syntology_url":"https://syntology.ai/paper/2306.13229","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.13229"}},"official":{"repos":["frankzheng2022/taco"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/jumanji-a-diverse-suite-of-scalable","slug":"jumanji-a-diverse-suite-of-scalable","title":"Jumanji: a Diverse Suite of Scalable Reinforcement Learning Environments in JAX","date":"2023-06-16","arxiv_id":"2306.09884","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/jumanji-a-diverse-suite-of-scalable#ran","syntology_url":"https://syntology.ai/paper/2306.09884","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.09884"}},"official":{"repos":["instadeepai/jumanji"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/datasets-and-benchmarks-for-offline-safe","slug":"datasets-and-benchmarks-for-offline-safe","title":"Datasets and Benchmarks for Offline Safe Reinforcement Learning","date":"2023-06-15","arxiv_id":"2306.09303","repositories_listed":3,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/datasets-and-benchmarks-for-offline-safe#ran","syntology_url":"https://syntology.ai/paper/2306.09303","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.09303"}},"official":{"repos":["liuzuxin/dsrl","liuzuxin/fsrl","liuzuxin/osrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/transcendental-idealism-of-planner-evaluating","slug":"transcendental-idealism-of-planner-evaluating","title":"Transcendental Idealism of Planner: Evaluating Perception from Planning Perspective for Autonomous Driving","date":"2023-06-12","arxiv_id":"2306.07276","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/transcendental-idealism-of-planner-evaluating#ran","syntology_url":"https://syntology.ai/paper/2306.07276","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.07276"}},"official":{"repos":["qcraftai/tip"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/policy-regularization-with-dataset-constraint","slug":"policy-regularization-with-dataset-constraint","title":"Policy Regularization with Dataset Constraint for Offline Reinforcement Learning","date":"2023-06-11","arxiv_id":"2306.06569","repositories_listed":2,"syntology":{"n":7,"n_ran":6,"n_constructed":6,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 6 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 6 samples that ran constructed an object rather than computing a result","sample_list":"/paper/policy-regularization-with-dataset-constraint#ran","syntology_url":"https://syntology.ai/paper/2306.06569","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.06569"}},"official":{"repos":["lamda-rl/prdc"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/generalizable-wireless-navigation-through","slug":"generalizable-wireless-navigation-through","title":"Digital Twin-Enhanced Wireless Indoor Navigation: Achieving Efficient Environment Sensing with Zero-Shot Reinforcement Learning","date":"2023-06-11","arxiv_id":"2306.06766","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generalizable-wireless-navigation-through#ran","syntology_url":"https://syntology.ai/paper/2306.06766","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.06766"}},"official":{"repos":["panshark/pirl-win"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/an-end-to-end-reinforcement-learning-approach","slug":"an-end-to-end-reinforcement-learning-approach","title":"An End-to-End Reinforcement Learning Approach for Job-Shop Scheduling Problems Based on Constraint Programming","date":"2023-06-09","arxiv_id":"2306.05747","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/an-end-to-end-reinforcement-learning-approach#ran","syntology_url":"https://syntology.ai/paper/2306.05747","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.05747"}},"official":{"repos":["ingambe/End2End-Job-Shop-Scheduling-CP"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/offline-prioritized-experience-replay","slug":"offline-prioritized-experience-replay","title":"Decoupled Prioritized Resampling for Offline RL","date":"2023-06-08","arxiv_id":"2306.05412","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/offline-prioritized-experience-replay#ran","syntology_url":"https://syntology.ai/paper/2306.05412","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.05412"}},"official":{"repos":["sail-sg/oper","yueyang130/odpr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/look-beneath-the-surface-exploiting-1","slug":"look-beneath-the-surface-exploiting-1","title":"Look Beneath the Surface: Exploiting Fundamental Symmetry for Sample-Efficient Offline RL","date":"2023-06-07","arxiv_id":"2306.04220","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":1,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/look-beneath-the-surface-exploiting-1#ran","syntology_url":"https://syntology.ai/paper/2306.04220","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.04220"}},"official":{"repos":["pcheng2/tsrl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/stabilizing-contrastive-rl-techniques-for","slug":"stabilizing-contrastive-rl-techniques-for","title":"Stabilizing Contrastive RL: Techniques for Robotic Goal Reaching from Offline Data","date":"2023-06-06","arxiv_id":"2306.03346","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/stabilizing-contrastive-rl-techniques-for#ran","syntology_url":"https://syntology.ai/paper/2306.03346","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.03346"}},"official":{"repos":["chongyi-zheng/stable_contrastive_rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/risk-aware-reward-shaping-of-reinforcement","slug":"risk-aware-reward-shaping-of-reinforcement","title":"Risk-Aware Reward Shaping of Reinforcement Learning Agents for Autonomous Driving","date":"2023-06-05","arxiv_id":"2306.03220","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/risk-aware-reward-shaping-of-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2306.03220","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.03220"}},"official":{"repos":["zhang-zengjie/code_2023_iecon_shaping_wu"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/for-sale-state-action-representation-learning-1","slug":"for-sale-state-action-representation-learning-1","title":"For SALE: State-Action Representation Learning for Deep Reinforcement Learning","date":"2023-06-04","arxiv_id":"2306.02451","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/for-sale-state-action-representation-learning-1#ran","syntology_url":"https://syntology.ai/paper/2306.02451","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.02451"}},"official":{"repos":["sfujim/td7"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/hyperparameters-in-reinforcement-learning-and","slug":"hyperparameters-in-reinforcement-learning-and","title":"Hyperparameters in Reinforcement Learning and How To Tune Them","date":"2023-06-02","arxiv_id":"2306.01324","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/hyperparameters-in-reinforcement-learning-and#ran","syntology_url":"https://syntology.ai/paper/2306.01324","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.01324"}},"official":{"repos":["facebookresearch/how-to-autorl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/tackling-unbounded-state-spaces-in-continuing","slug":"tackling-unbounded-state-spaces-in-continuing","title":"Learning to Stabilize Online Reinforcement Learning in Unbounded State Spaces","date":"2023-06-02","arxiv_id":"2306.01896","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":1,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":2,"n_no_contract":1,"n_pointer_only":6,"phrase":"6 ran (of which 1 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 2 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tackling-unbounded-state-spaces-in-continuing#ran","syntology_url":"https://syntology.ai/paper/2306.01896","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.01896"}},"official":{"repos":["badger-rl/stop"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":1,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/thought-cloning-learning-to-think-while-1","slug":"thought-cloning-learning-to-think-while-1","title":"Thought Cloning: Learning to Think while Acting by Imitating Human Thinking","date":"2023-06-01","arxiv_id":"2306.00323","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/thought-cloning-learning-to-think-while-1#ran","syntology_url":"https://syntology.ai/paper/2306.00323","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.00323"}},"official":{"repos":["ShengranHu/Thought-Cloning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/torchrl-a-data-driven-decision-making-library","slug":"torchrl-a-data-driven-decision-making-library","title":"TorchRL: A data-driven decision-making library for PyTorch","date":"2023-06-01","arxiv_id":"2306.00577","repositories_listed":2,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/torchrl-a-data-driven-decision-making-library#ran","syntology_url":"https://syntology.ai/paper/2306.00577","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.00577"}},"official":null}},{"url":"/paper/identifiability-and-generalizability-in","slug":"identifiability-and-generalizability-in","title":"Identifiability and Generalizability in Constrained Inverse Reinforcement Learning","date":"2023-06-01","arxiv_id":"2306.00629","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/identifiability-and-generalizability-in#ran","syntology_url":"https://syntology.ai/paper/2306.00629","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.00629"}},"official":{"repos":["andrschl/cirl"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/normalization-enhances-generalization-in","slug":"normalization-enhances-generalization-in","title":"Normalization Enhances Generalization in Visual Reinforcement Learning","date":"2023-06-01","arxiv_id":"2306.00656","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/normalization-enhances-generalization-in#ran","syntology_url":"https://syntology.ai/paper/2306.00656","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.00656"}},"official":{"repos":["lilucse/Normalization-Enhances-Generalization-in-Visual-Reinforcement-Learning"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/let-s-verify-step-by-step-1","slug":"let-s-verify-step-by-step-1","title":"Let's Verify Step by Step","date":"2023-05-31","arxiv_id":"2305.20050","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":2,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 2 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/let-s-verify-step-by-step-1#ran","syntology_url":"https://syntology.ai/paper/2305.20050","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.20050"}},"official":{"repos":["openai/prm800k"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-diffusion-policies-for-offline-1","slug":"efficient-diffusion-policies-for-offline-1","title":"Efficient Diffusion Policies for Offline Reinforcement Learning","date":"2023-05-31","arxiv_id":"2305.20081","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-diffusion-policies-for-offline-1#ran","syntology_url":"https://syntology.ai/paper/2305.20081","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.20081"}},"official":{"repos":["sail-sg/edp"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-for-edge-weighted-online-bipartite","slug":"learning-for-edge-weighted-online-bipartite","title":"Learning for Edge-Weighted Online Bipartite Matching with Robustness Guarantees","date":"2023-05-31","arxiv_id":"2306.00172","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":3,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","sample_list":"/paper/learning-for-edge-weighted-online-bipartite#ran","syntology_url":"https://syntology.ai/paper/2306.00172","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.00172"}},"official":{"repos":["ren-research/lomar"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/subequivariant-graph-reinforcement-learning","slug":"subequivariant-graph-reinforcement-learning","title":"Subequivariant Graph Reinforcement Learning in 3D Environments","date":"2023-05-30","arxiv_id":"2305.18951","repositories_listed":1,"syntology":{"n":18,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":13,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 13 unverified","sample_list":"/paper/subequivariant-graph-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2305.18951","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.18951"}},"official":{"repos":["alpc91/sgrl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":13,"ran_from_kinds":["official"]}}},{"url":"/paper/provable-and-practical-efficient-exploration","slug":"provable-and-practical-efficient-exploration","title":"Provable and Practical: Efficient Exploration in Reinforcement Learning via Langevin Monte Carlo","date":"2023-05-29","arxiv_id":"2305.18246","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/provable-and-practical-efficient-exploration#ran","syntology_url":"https://syntology.ai/paper/2305.18246","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.18246"}},"official":{"repos":["hmishfaq/lmc-lsvi"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/diffusion-model-is-an-effective-planner-and-1","slug":"diffusion-model-is-an-effective-planner-and-1","title":"Diffusion Model is an Effective Planner and Data Synthesizer for Multi-Task Reinforcement Learning","date":"2023-05-29","arxiv_id":"2305.18459","repositories_listed":1,"syntology":{"n":30,"n_ran":19,"n_constructed":3,"n_ran_checked":18,"n_instrument":1,"n_unverified":11,"n_honours":5,"n_violates":0,"n_no_contract":13,"n_pointer_only":4,"phrase":"19 ran (of which 3 constructed an object rather than computing a result; 18 with no instrument failure: 5 honoured, 0 violated, 13 with no contract checked; 1 where Syntology's instrument failed) · 11 unverified","sample_list":"/paper/diffusion-model-is-an-effective-planner-and-1#ran","syntology_url":"https://syntology.ai/paper/2305.18459","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.18459"}},"official":{"repos":["tinnerhrhe/MTDiff"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":7,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/madiff-offline-multi-agent-learning-with","slug":"madiff-offline-multi-agent-learning-with","title":"MADiff: Offline Multi-agent Learning with Diffusion Models","date":"2023-05-27","arxiv_id":"2305.17330","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/madiff-offline-multi-agent-learning-with#ran","syntology_url":"https://syntology.ai/paper/2305.17330","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.17330"}},"official":{"repos":["zbzhu99/madiff"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/future-conditioned-unsupervised-pretraining","slug":"future-conditioned-unsupervised-pretraining","title":"Future-conditioned Unsupervised Pretraining for Decision Transformer","date":"2023-05-26","arxiv_id":"2305.16683","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/future-conditioned-unsupervised-pretraining#ran","syntology_url":"https://syntology.ai/paper/2305.16683","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.16683"}},"official":{"repos":["fffffarmer/pdt"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/decision-aware-actor-critic-with-function-1","slug":"decision-aware-actor-critic-with-function-1","title":"Decision-Aware Actor-Critic with Function Approximation and Theoretical Guarantees","date":"2023-05-24","arxiv_id":"2305.15249","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/decision-aware-actor-critic-with-function-1#ran","syntology_url":"https://syntology.ai/paper/2305.15249","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.15249"}},"official":{"repos":["amirrezakazemi/acpg"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/collaborative-world-models-an-online-offline","slug":"collaborative-world-models-an-online-offline","title":"Making Offline RL Online: Collaborative World Models for Offline Visual Reinforcement Learning","date":"2023-05-24","arxiv_id":"2305.15260","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":7,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":13,"phrase":"10 ran (of which 7 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/collaborative-world-models-an-online-offline#ran","syntology_url":"https://syntology.ai/paper/2305.15260","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.15260"}},"official":{"repos":["qiwang067/CoWorld"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":7,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/conditional-mutual-information-for-1","slug":"conditional-mutual-information-for-1","title":"Conditional Mutual Information for Disentangled Representations in Reinforcement Learning","date":"2023-05-23","arxiv_id":"2305.14133","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/conditional-mutual-information-for-1#ran","syntology_url":"https://syntology.ai/paper/2305.14133","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14133"}},"official":{"repos":["uoe-agents/cmid"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/2305-14550","slug":"2305-14550","title":"When should we prefer Decision Transformers for Offline Reinforcement Learning?","date":"2023-05-23","arxiv_id":"2305.14550","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/2305-14550#ran","syntology_url":"https://syntology.ai/paper/2305.14550","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14550"}},"official":{"repos":["prajjwal1/rl_paradigm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/furniturebench-reproducible-real-world","slug":"furniturebench-reproducible-real-world","title":"FurnitureBench: Reproducible Real-World Benchmark for Long-Horizon Complex Manipulation","date":"2023-05-22","arxiv_id":"2305.12821","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/furniturebench-reproducible-real-world#ran","syntology_url":"https://syntology.ai/paper/2305.12821","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.12821"}},"official":{"repos":["clvrai/furniture-bench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/policy-representation-via-diffusion","slug":"policy-representation-via-diffusion","title":"Policy Representation via Diffusion Probability Model for Reinforcement Learning","date":"2023-05-22","arxiv_id":"2305.13122","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/policy-representation-via-diffusion#ran","syntology_url":"https://syntology.ai/paper/2305.13122","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.13122"}},"official":{"repos":["bellmantimehut/dipo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-diverse-risk-preferences-in","slug":"learning-diverse-risk-preferences-in","title":"Learning Diverse Risk Preferences in Population-based Self-play","date":"2023-05-19","arxiv_id":"2305.11476","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-diverse-risk-preferences-in#ran","syntology_url":"https://syntology.ai/paper/2305.11476","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.11476"}},"official":{"repos":["jackory/rpbt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/demonstration-free-autonomous-reinforcement","slug":"demonstration-free-autonomous-reinforcement","title":"Demonstration-free Autonomous Reinforcement Learning via Implicit and Bidirectional Curriculum","date":"2023-05-17","arxiv_id":"2305.09943","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":1,"n_ran_checked":3,"n_instrument":3,"n_unverified":2,"n_honours":2,"n_violates":1,"n_no_contract":0,"n_pointer_only":8,"phrase":"6 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/demonstration-free-autonomous-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2305.09943","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.09943"}},"official":{"repos":["snu-larr/ibc_official"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/revisiting-the-minimalist-approach-to-offline","slug":"revisiting-the-minimalist-approach-to-offline","title":"Revisiting the Minimalist Approach to Offline Reinforcement Learning","date":"2023-05-16","arxiv_id":"2305.09836","repositories_listed":3,"syntology":{"n":16,"n_ran":15,"n_constructed":0,"n_ran_checked":13,"n_instrument":2,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":11,"n_pointer_only":1,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 2 honoured, 0 violated, 11 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/revisiting-the-minimalist-approach-to-offline#ran","syntology_url":"https://syntology.ai/paper/2305.09836","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.09836"}},"official":{"repos":["dt6a/rebrac"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/towards-generalizable-reinforcement-learning","slug":"towards-generalizable-reinforcement-learning","title":"Towards Generalizable Reinforcement Learning for Trade Execution","date":"2023-05-12","arxiv_id":"2307.11685","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/towards-generalizable-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2307.11685","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.11685"}},"official":null}},{"url":"/paper/explaining-rl-decisions-with-trajectories","slug":"explaining-rl-decisions-with-trajectories","title":"Explaining RL Decisions with Trajectories","date":"2023-05-06","arxiv_id":"2305.04073","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/explaining-rl-decisions-with-trajectories#ran","syntology_url":"https://syntology.ai/paper/2305.04073","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.04073"}},"official":{"repos":["shripaddeshmukh/xrl_with_trajectories"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/leveraging-factored-action-spaces-for","slug":"leveraging-factored-action-spaces-for","title":"Leveraging Factored Action Spaces for Efficient Offline Reinforcement Learning in Healthcare","date":"2023-05-02","arxiv_id":"2305.01738","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/leveraging-factored-action-spaces-for#ran","syntology_url":"https://syntology.ai/paper/2305.01738","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.01738"}},"official":{"repos":["mld3/offlinerl_factoredactions"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/distance-weighted-supervised-learning-for","slug":"distance-weighted-supervised-learning-for","title":"Distance Weighted Supervised Learning for Offline Interaction Data","date":"2023-04-26","arxiv_id":"2304.13774","repositories_listed":1,"syntology":{"n":27,"n_ran":14,"n_constructed":0,"n_ran_checked":4,"n_instrument":10,"n_unverified":13,"n_honours":2,"n_violates":1,"n_no_contract":1,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 2 honoured, 1 violated, 1 with no contract checked; 10 where Syntology's instrument failed) · 13 unverified","sample_list":"/paper/distance-weighted-supervised-learning-for#ran","syntology_url":"https://syntology.ai/paper/2304.13774","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.13774"}},"official":null}},{"url":"/paper/contrastive-energy-prediction-for-exact","slug":"contrastive-energy-prediction-for-exact","title":"Contrastive Energy Prediction for Exact Energy-Guided Diffusion Sampling in Offline Reinforcement Learning","date":"2023-04-25","arxiv_id":"2304.12824","repositories_listed":3,"syntology":{"n":27,"n_ran":19,"n_constructed":1,"n_ran_checked":13,"n_instrument":6,"n_unverified":8,"n_honours":3,"n_violates":0,"n_no_contract":10,"n_pointer_only":17,"phrase":"19 ran (of which 1 constructed an object rather than computing a result; 13 with no instrument failure: 3 honoured, 0 violated, 10 with no contract checked; 6 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/contrastive-energy-prediction-for-exact#ran","syntology_url":"https://syntology.ai/paper/2304.12824","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.12824"}},"official":{"repos":["thu-ml/cep-energy-guided-diffusion"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":6,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/benchmarking-actor-critic-deep-reinforcement","slug":"benchmarking-actor-critic-deep-reinforcement","title":"Benchmarking Actor-Critic Deep Reinforcement Learning Algorithms for Robotics Control with Action Constraints","date":"2023-04-18","arxiv_id":"2304.08743","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-actor-critic-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2304.08743","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.08743"}},"official":{"repos":["omron-sinicx/action-constrained-rl-benchmark"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/autorl-hyperparameter-landscapes","slug":"autorl-hyperparameter-landscapes","title":"AutoRL Hyperparameter Landscapes","date":"2023-04-05","arxiv_id":"2304.02396","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/autorl-hyperparameter-landscapes#ran","syntology_url":"https://syntology.ai/paper/2304.02396","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.02396"}},"official":{"repos":["automl/autorl-landscape"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/optimal-goal-reaching-reinforcement-learning","slug":"optimal-goal-reaching-reinforcement-learning","title":"Optimal Goal-Reaching Reinforcement Learning via Quasimetric Learning","date":"2023-04-03","arxiv_id":"2304.01203","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/optimal-goal-reaching-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2304.01203","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.01203"}},"official":{"repos":["quasimetric-learning/quasimetric-rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mahalo-unifying-offline-reinforcement","slug":"mahalo-unifying-offline-reinforcement","title":"MAHALO: Unifying Offline Reinforcement Learning and Imitation Learning from Observations","date":"2023-03-30","arxiv_id":"2303.17156","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mahalo-unifying-offline-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2303.17156","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.17156"}},"official":{"repos":["anqili/mahalo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/language-models-can-solve-computer-tasks","slug":"language-models-can-solve-computer-tasks","title":"Language Models can Solve Computer Tasks","date":"2023-03-30","arxiv_id":"2303.17491","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/language-models-can-solve-computer-tasks#ran","syntology_url":"https://syntology.ai/paper/2303.17491","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.17491"}},"official":{"repos":["posgnu/rci-agent"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/offline-rl-with-no-ood-actions-in-sample","slug":"offline-rl-with-no-ood-actions-in-sample","title":"Offline RL with No OOD Actions: In-Sample Learning via Implicit Value Regularization","date":"2023-03-28","arxiv_id":"2303.15810","repositories_listed":4,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/offline-rl-with-no-ood-actions-in-sample#ran","syntology_url":"https://syntology.ai/paper/2303.15810","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.15810"}},"official":{"repos":["ryanxhr/ivr"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/optimal-transport-for-offline-imitation","slug":"optimal-transport-for-offline-imitation","title":"Optimal Transport for Offline Imitation Learning","date":"2023-03-24","arxiv_id":"2303.13971","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/optimal-transport-for-offline-imitation#ran","syntology_url":"https://syntology.ai/paper/2303.13971","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.13971"}},"official":{"repos":["ethanluoyc/optimal_transport_reward"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/imitating-graph-based-planning-with-goal","slug":"imitating-graph-based-planning-with-goal","title":"Imitating Graph-Based Planning with Goal-Conditioned Policies","date":"2023-03-20","arxiv_id":"2303.11166","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/imitating-graph-based-planning-with-goal#ran","syntology_url":"https://syntology.ai/paper/2303.11166","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.11166"}},"official":{"repos":["junsu-kim97/pig"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/clip4mc-an-rl-friendly-vision-language-model","slug":"clip4mc-an-rl-friendly-vision-language-model","title":"Reinforcement Learning Friendly Vision-Language Model for Minecraft","date":"2023-03-19","arxiv_id":"2303.10571","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/clip4mc-an-rl-friendly-vision-language-model#ran","syntology_url":"https://syntology.ai/paper/2303.10571","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.10571"}},"official":{"repos":["PKU-RL/CLIP4MC"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dynamic-update-to-data-ratio-minimizing-world","slug":"dynamic-update-to-data-ratio-minimizing-world","title":"Dynamic Update-to-Data Ratio: Minimizing World Model Overfitting","date":"2023-03-17","arxiv_id":"2303.10144","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dynamic-update-to-data-ratio-minimizing-world#ran","syntology_url":"https://syntology.ai/paper/2303.10144","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.10144"}},"official":{"repos":["nicolinho/dutd"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/kernel-density-bayesian-inverse-reinforcement","slug":"kernel-density-bayesian-inverse-reinforcement","title":"Kernel Density Bayesian Inverse Reinforcement Learning","date":"2023-03-13","arxiv_id":"2303.06827","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/kernel-density-bayesian-inverse-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2303.06827","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.06827"}},"official":{"repos":["bee-hive/kdbirl_public"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/transformer-based-world-models-are-happy-with","slug":"transformer-based-world-models-are-happy-with","title":"Transformer-based World Models Are Happy With 100k Interactions","date":"2023-03-13","arxiv_id":"2303.07109","repositories_listed":1,"syntology":{"n":25,"n_ran":16,"n_constructed":6,"n_ran_checked":8,"n_instrument":8,"n_unverified":9,"n_honours":1,"n_violates":1,"n_no_contract":6,"n_pointer_only":0,"phrase":"16 ran (of which 6 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 1 violated, 6 with no contract checked; 8 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/transformer-based-world-models-are-happy-with#ran","syntology_url":"https://syntology.ai/paper/2303.07109","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.07109"}},"official":{"repos":["jrobine/twm"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":6,"n_ran_no_instrument_failure":8,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/synthetic-experience-replay-1","slug":"synthetic-experience-replay-1","title":"Synthetic Experience Replay","date":"2023-03-12","arxiv_id":"2303.06614","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":1,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/synthetic-experience-replay-1#ran","syntology_url":"https://syntology.ai/paper/2303.06614","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.06614"}},"official":{"repos":["conglu1997/SynthER"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/mcts-geb-monte-carlo-tree-search-is-a-good-e","slug":"mcts-geb-monte-carlo-tree-search-is-a-good-e","title":"MCTS-GEB: Monte Carlo Tree Search is a Good E-graph Builder","date":"2023-03-08","arxiv_id":"2303.04651","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mcts-geb-monte-carlo-tree-search-is-a-good-e#ran","syntology_url":"https://syntology.ai/paper/2303.04651","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.04651"}},"official":{"repos":["ucamrl/eqs"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/zeroth-order-optimization-meets-human","slug":"zeroth-order-optimization-meets-human","title":"Zeroth-Order Optimization Meets Human Feedback: Provable Learning via Ranking Oracles","date":"2023-03-07","arxiv_id":"2303.03751","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/zeroth-order-optimization-meets-human#ran","syntology_url":"https://syntology.ai/paper/2303.03751","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.03751"}},"official":{"repos":["TZW1998/Taming-Stable-Diffusion-with-Human-Ranking-Feedback"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/diminishing-return-of-value-expansion-methods","slug":"diminishing-return-of-value-expansion-methods","title":"Diminishing Return of Value Expansion Methods in Model-Based Reinforcement Learning","date":"2023-03-07","arxiv_id":"2303.03955","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/diminishing-return-of-value-expansion-methods#ran","syntology_url":"https://syntology.ai/paper/2303.03955","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.03955"}},"official":{"repos":["danielpalen/value_expansion"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/safe-reinforcement-learning-via-probabilistic","slug":"safe-reinforcement-learning-via-probabilistic","title":"Safe Reinforcement Learning via Probabilistic Logic Shields","date":"2023-03-06","arxiv_id":"2303.03226","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/safe-reinforcement-learning-via-probabilistic#ran","syntology_url":"https://syntology.ai/paper/2303.03226","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.03226"}},"official":null}},{"url":"/paper/neural-airport-ground-handling","slug":"neural-airport-ground-handling","title":"Neural Airport Ground Handling","date":"2023-03-04","arxiv_id":"2303.02442","repositories_listed":1,"syntology":{"n":10,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/neural-airport-ground-handling#ran","syntology_url":"https://syntology.ai/paper/2303.02442","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.02442"}},"official":{"repos":["royalskye/agh"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/preference-transformer-modeling-human","slug":"preference-transformer-modeling-human","title":"Preference Transformer: Modeling Human Preferences using Transformers for RL","date":"2023-03-02","arxiv_id":"2303.00957","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":6,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 1 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/preference-transformer-modeling-human#ran","syntology_url":"https://syntology.ai/paper/2303.00957","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.00957"}},"official":{"repos":["csmile-1006/preferencetransformer"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-control-autonomous-fleets-from","slug":"learning-to-control-autonomous-fleets-from","title":"Learning to Control Autonomous Fleets from Observation via Offline Reinforcement Learning","date":"2023-02-28","arxiv_id":"2302.14833","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-to-control-autonomous-fleets-from#ran","syntology_url":"https://syntology.ai/paper/2302.14833","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.14833"}},"official":{"repos":["carolinssc/offline-rl-amod","carolinssc/offline-rl-for-amod"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/systematic-rectification-of-language-models","slug":"systematic-rectification-of-language-models","title":"Systematic Rectification of Language Models via Dead-end Analysis","date":"2023-02-27","arxiv_id":"2302.14003","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/systematic-rectification-of-language-models#ran","syntology_url":"https://syntology.ai/paper/2302.14003","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.14003"}},"official":{"repos":["mcao516/rectification-lm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/model-based-uncertainty-in-value-functions","slug":"model-based-uncertainty-in-value-functions","title":"Model-Based Uncertainty in Value Functions","date":"2023-02-24","arxiv_id":"2302.12526","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/model-based-uncertainty-in-value-functions#ran","syntology_url":"https://syntology.ai/paper/2302.12526","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.12526"}},"official":{"repos":["boschresearch/ube-mbrl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/neural-laplace-control-for-continuous-time","slug":"neural-laplace-control-for-continuous-time","title":"Neural Laplace Control for Continuous-time Delayed Systems","date":"2023-02-24","arxiv_id":"2302.12604","repositories_listed":2,"syntology":{"n":17,"n_ran":13,"n_constructed":0,"n_ran_checked":9,"n_instrument":4,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":8,"n_pointer_only":1,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 1 violated, 8 with no contract checked; 4 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/neural-laplace-control-for-continuous-time#ran","syntology_url":"https://syntology.ai/paper/2302.12604","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.12604"}},"official":{"repos":["samholt/neurallaplacecontrol","vanderschaarlab/neurallaplacecontrol"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/the-dormant-neuron-phenomenon-in-deep","slug":"the-dormant-neuron-phenomenon-in-deep","title":"The Dormant Neuron Phenomenon in Deep Reinforcement Learning","date":"2023-02-24","arxiv_id":"2302.12902","repositories_listed":3,"syntology":{"n":8,"n_ran":4,"n_constructed":1,"n_ran_checked":1,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":8,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/the-dormant-neuron-phenomenon-in-deep#ran","syntology_url":"https://syntology.ai/paper/2302.12902","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.12902"}},"official":{"repos":["google/dopamine"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/behavior-proximal-policy-optimization","slug":"behavior-proximal-policy-optimization","title":"Behavior Proximal Policy Optimization","date":"2023-02-22","arxiv_id":"2302.11312","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/behavior-proximal-policy-optimization#ran","syntology_url":"https://syntology.ai/paper/2302.11312","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.11312"}},"official":{"repos":["dragon-zhuang/bppo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fantastic-rewards-and-how-to-tame-them-a-case","slug":"fantastic-rewards-and-how-to-tame-them-a-case","title":"Fantastic Rewards and How to Tame Them: A Case Study on Reward Learning for Task-oriented Dialogue Systems","date":"2023-02-20","arxiv_id":"2302.10342","repositories_listed":2,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/fantastic-rewards-and-how-to-tame-them-a-case#ran","syntology_url":"https://syntology.ai/paper/2302.10342","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.10342"}},"official":{"repos":["budzianowski/multiwoz","shentao-yang/fantastic_reward_iclr2023"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/imitation-from-arbitrary-experience-a-dual","slug":"imitation-from-arbitrary-experience-a-dual","title":"Dual RL: Unification and New Methods for Reinforcement and Imitation Learning","date":"2023-02-16","arxiv_id":"2302.08560","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/imitation-from-arbitrary-experience-a-dual#ran","syntology_url":"https://syntology.ai/paper/2302.08560","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.08560"}},"official":{"repos":["hari-sikchi/DVL"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-deep-reinforcement-learning-through-2","slug":"robust-deep-reinforcement-learning-through-2","title":"Regret-Based Defense in Adversarial Reinforcement Learning","date":"2023-02-14","arxiv_id":"2302.06912","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":6,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/robust-deep-reinforcement-learning-through-2#ran","syntology_url":"https://syntology.ai/paper/2302.06912","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.06912"}},"official":{"repos":["romanbelaire/robust-ccer"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/constrained-decision-transformer-for-offline","slug":"constrained-decision-transformer-for-offline","title":"Constrained Decision Transformer for Offline Safe Reinforcement Learning","date":"2023-02-14","arxiv_id":"2302.07351","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":2,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/constrained-decision-transformer-for-offline#ran","syntology_url":"https://syntology.ai/paper/2302.07351","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.07351"}},"official":{"repos":["liuzuxin/osrl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/automatic-noise-filtering-with-dynamic-sparse","slug":"automatic-noise-filtering-with-dynamic-sparse","title":"Automatic Noise Filtering with Dynamic Sparse Training in Deep Reinforcement Learning","date":"2023-02-13","arxiv_id":"2302.06548","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/automatic-noise-filtering-with-dynamic-sparse#ran","syntology_url":"https://syntology.ai/paper/2302.06548","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.06548"}},"official":{"repos":["bramgrooten/automatic-noise-filtering"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-penalty-based-bilevel-gradient-descent","slug":"on-penalty-based-bilevel-gradient-descent","title":"On Penalty-based Bilevel Gradient Descent Method","date":"2023-02-10","arxiv_id":"2302.05185","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":3,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 3 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/on-penalty-based-bilevel-gradient-descent#ran","syntology_url":"https://syntology.ai/paper/2302.05185","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.05185"}},"official":{"repos":["hanshen95/penalized-bilevel-gradient-descent"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hierarchical-generative-adversarial-imitation","slug":"hierarchical-generative-adversarial-imitation","title":"Hierarchical Generative Adversarial Imitation Learning with Mid-level Input Generation for Autonomous Driving on Urban Environments","date":"2023-02-09","arxiv_id":"2302.04823","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hierarchical-generative-adversarial-imitation#ran","syntology_url":"https://syntology.ai/paper/2302.04823","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.04823"}},"official":{"repos":["gustavokcouto/hgail"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-task-recommendations-with-reinforcement","slug":"multi-task-recommendations-with-reinforcement","title":"Multi-Task Recommendations with Reinforcement Learning","date":"2023-02-07","arxiv_id":"2302.03328","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-task-recommendations-with-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2302.03328","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.03328"}},"official":{"repos":["applied-machine-learning-lab/rmtl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/grounding-large-language-models-in","slug":"grounding-large-language-models-in","title":"Grounding Large Language Models in Interactive Environments with Online Reinforcement Learning","date":"2023-02-06","arxiv_id":"2302.02662","repositories_listed":3,"syntology":{"n":12,"n_ran":9,"n_constructed":2,"n_ran_checked":6,"n_instrument":3,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"9 ran (of which 2 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/grounding-large-language-models-in#ran","syntology_url":"https://syntology.ai/paper/2302.02662","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.02662"}},"official":{"repos":["clementromac/lamorel","flowersteam/grounding_llms_with_online_rl","flowersteam/lamorel"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":2,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-online-reinforcement-learning-with","slug":"efficient-online-reinforcement-learning-with","title":"Efficient Online Reinforcement Learning with Offline Data","date":"2023-02-06","arxiv_id":"2302.02948","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-online-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2302.02948","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.02948"}},"official":{"repos":["ikostrikov/rlpd"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/locally-constrained-policy-optimization-for","slug":"locally-constrained-policy-optimization-for","title":"Online Reinforcement Learning in Non-Stationary Context-Driven Environments","date":"2023-02-04","arxiv_id":"2302.02182","repositories_listed":2,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/locally-constrained-policy-optimization-for#ran","syntology_url":"https://syntology.ai/paper/2302.02182","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.02182"}},"official":{"repos":["lcpo-rl/lcpo","pouyahmdn/lcpo"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-optimize-for-reinforcement","slug":"learning-to-optimize-for-reinforcement","title":"Learning to Optimize for Reinforcement Learning","date":"2023-02-03","arxiv_id":"2302.01470","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-to-optimize-for-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2302.01470","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.01470"}},"official":{"repos":["sail-sg/optim4rl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/average-constrained-policy-optimization","slug":"average-constrained-policy-optimization","title":"ACPO: A Policy Optimization Algorithm for Average MDPs with Constraints","date":"2023-02-02","arxiv_id":"2302.00808","repositories_listed":0,"syntology":{"n":12,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/average-constrained-policy-optimization#ran","syntology_url":"https://syntology.ai/paper/2302.00808","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.00808"}},"official":null}},{"url":"/paper/policy-expansion-for-bridging-offline-to","slug":"policy-expansion-for-bridging-offline-to","title":"Policy Expansion for Bridging Offline-to-Online Reinforcement Learning","date":"2023-02-02","arxiv_id":"2302.00935","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/policy-expansion-for-bridging-offline-to#ran","syntology_url":"https://syntology.ai/paper/2302.00935","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.00935"}},"official":{"repos":["haichao-zhang/pex"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/internally-rewarded-reinforcement-learning","slug":"internally-rewarded-reinforcement-learning","title":"Internally Rewarded Reinforcement Learning","date":"2023-02-01","arxiv_id":"2302.00270","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/internally-rewarded-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2302.00270","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.00270"}},"official":{"repos":["mengdi-li/internally-rewarded-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/off-the-grid-marl-a-framework-for-dataset","slug":"off-the-grid-marl-a-framework-for-dataset","title":"Off-the-Grid MARL: Datasets with Baselines for Offline Multi-Agent Reinforcement Learning","date":"2023-02-01","arxiv_id":"2302.00521","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/off-the-grid-marl-a-framework-for-dataset#ran","syntology_url":"https://syntology.ai/paper/2302.00521","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.00521"}},"official":{"repos":["instadeepai/og-marl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-multi-task-reinforcement-learning","slug":"efficient-multi-task-reinforcement-learning","title":"QMP: Q-switch Mixture of Policies for Multi-Task Behavior Sharing","date":"2023-02-01","arxiv_id":"2302.00671","repositories_listed":0,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 1 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-multi-task-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2302.00671","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.00671"}},"official":null}},{"url":"/paper/collaborating-with-language-models-for","slug":"collaborating-with-language-models-for","title":"Collaborating with language models for embodied reasoning","date":"2023-02-01","arxiv_id":"2302.00763","repositories_listed":0,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/collaborating-with-language-models-for#ran","syntology_url":"https://syntology.ai/paper/2302.00763","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.00763"}},"official":null}},{"url":"/paper/optimizing-ddpm-sampling-with-shortcut-fine","slug":"optimizing-ddpm-sampling-with-shortcut-fine","title":"Optimizing DDPM Sampling with Shortcut Fine-Tuning","date":"2023-01-31","arxiv_id":"2301.13362","repositories_listed":1,"syntology":{"n":9,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":5,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/optimizing-ddpm-sampling-with-shortcut-fine#ran","syntology_url":"https://syntology.ai/paper/2301.13362","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.13362"}},"official":{"repos":["uw-madison-lee-lab/sft-pg"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/retrosynthetic-planning-with-dual-value","slug":"retrosynthetic-planning-with-dual-value","title":"Retrosynthetic Planning with Dual Value Networks","date":"2023-01-31","arxiv_id":"2301.13755","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":2,"n_ran_checked":2,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":8,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/retrosynthetic-planning-with-dual-value#ran","syntology_url":"https://syntology.ai/paper/2301.13755","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.13755"}},"official":{"repos":["DiXue98/PDVN"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/execution-based-code-generation-using-deep","slug":"execution-based-code-generation-using-deep","title":"Execution-based Code Generation using Deep Reinforcement Learning","date":"2023-01-31","arxiv_id":"2301.13816","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/execution-based-code-generation-using-deep#ran","syntology_url":"https://syntology.ai/paper/2301.13816","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.13816"}},"official":{"repos":["reddy-lab-code-research/PPOCoder"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"5b706d9dbd751fc216103d1971da5249d95a20cf0143b8bb9905fa3a148abeb9","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}